{ "best_global_step": 3100, "best_metric": 0.7422462472548852, "best_model_checkpoint": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_223409/checkpoint-3100", "epoch": 0.13254608989034825, "eval_steps": 50, "global_step": 3300, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "clip_ratio/high_max": 0.1411083247512579, "clip_ratio/high_mean": 0.1411083247512579, "clip_ratio/low_mean": 0.05967577267438173, "clip_ratio/low_min": 0.05967577267438173, "clip_ratio/region_mean": 0.20078409742563963, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 86.375, "completions/mean_terminated_length": 86.375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 1.9424520283937454, "epoch": 3.861600247142416e-05, "frac_reward_zero_std": 0.0, "grad_norm": 12.506416320800781, "learning_rate": 1e-05, "loss": -0.0412, "num_tokens": 2187.0, "reward": 0.46589815616607666, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_meter_mean": 0.5479353666305542, "reward_meter_std": 0.42796453833580017, "reward_std": 0.35266923904418945, "reward_total_composite_mean": 0.46589815616607666, "reward_total_composite_std": 0.35266923904418945, "reward_total_mean": 0.46589815616607666, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/meter/mean": 0.5479353666305542, "rewards/meter/std": 0.42796453833580017, "rewards/total_composite/mean": 0.46589815616607666, "rewards/total_composite/std": 0.35266923904418945, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005318284034729, "sampling/importance_sampling_ratio/min": 0.20743781328201294, "sampling/sampling_logp_difference/max": 1.5729236602783203, "sampling/sampling_logp_difference/mean": 0.20873695611953735, "step": 1 }, { "clip_ratio/high_max": 0.05052456725388765, "clip_ratio/high_mean": 0.05052456725388765, "clip_ratio/low_mean": 0.10001249238848686, "clip_ratio/low_min": 0.10001249238848686, "clip_ratio/region_mean": 0.15053705964237452, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 57.75, "completions/mean_terminated_length": 57.75, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.9319793060421944, "epoch": 7.723200494284832e-05, "frac_reward_zero_std": 0.0, "grad_norm": 17.5363826751709, "learning_rate": 9.996969696969698e-06, "loss": 0.0621, "num_tokens": 4001.0, "reward": 0.6565604209899902, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6565604209899902, "reward_meter_std": 0.3700096607208252, "reward_std": 0.3700096607208252, "reward_total_composite_mean": 0.6565604209899902, "reward_total_composite_std": 0.3700096607208252, "reward_total_mean": 0.6565604209899902, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6565604209899902, "rewards/meter/std": 0.3700096607208252, "rewards/total_composite/mean": 0.6565604209899902, "rewards/total_composite/std": 0.3700096607208252, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0128601789474487, "sampling/importance_sampling_ratio/min": 0.1371382474899292, "sampling/sampling_logp_difference/max": 1.986765742301941, "sampling/sampling_logp_difference/mean": 0.16318750381469727, "step": 2 }, { "clip_ratio/high_max": 0.10511028952896595, "clip_ratio/high_mean": 0.10511028952896595, "clip_ratio/low_mean": 0.11367196403443813, "clip_ratio/low_min": 0.11367196403443813, "clip_ratio/region_mean": 0.21878225356340408, "completions/clipped_ratio": 0.0, "completions/max_length": 46.0, "completions/max_terminated_length": 46.0, "completions/mean_length": 41.75, "completions/mean_terminated_length": 41.75, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 1.6740138679742813, "epoch": 0.00011584800741427248, "frac_reward_zero_std": 0.0, "grad_norm": 17.113204956054688, "learning_rate": 9.993939393939395e-06, "loss": 0.0587, "num_tokens": 5663.0, "reward": 0.5669770240783691, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5669770240783691, "reward_meter_std": 0.42305904626846313, "reward_std": 0.42305904626846313, "reward_total_composite_mean": 0.5669770240783691, "reward_total_composite_std": 0.42305904626846313, "reward_total_mean": 0.5669770240783691, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5669770240783691, "rewards/meter/std": 0.42305904626846313, "rewards/total_composite/mean": 0.5669770240783691, "rewards/total_composite/std": 0.42305904626846313, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.04336678981781, "sampling/importance_sampling_ratio/min": 0.23251527547836304, "sampling/sampling_logp_difference/max": 1.4587993621826172, "sampling/sampling_logp_difference/mean": 0.1992362141609192, "step": 3 }, { "clip_ratio/high_max": 0.15018654288724065, "clip_ratio/high_mean": 0.15018654288724065, "clip_ratio/low_mean": 0.029411764815449715, "clip_ratio/low_min": 0.029411764815449715, "clip_ratio/region_mean": 0.17959830770269036, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 26.75, "completions/mean_terminated_length": 26.75, "completions/min_length": 17.0, "completions/min_terminated_length": 17.0, "entropy": 1.2625475600361824, "epoch": 0.00015446400988569664, "frac_reward_zero_std": 0.0, "grad_norm": 19.642213821411133, "learning_rate": 9.990909090909093e-06, "loss": -0.1025, "num_tokens": 7109.0, "reward": 0.856600284576416, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.856600284576416, "reward_meter_std": 0.34479716420173645, "reward_std": 0.34479716420173645, "reward_total_composite_mean": 0.856600284576416, "reward_total_composite_std": 0.34479716420173645, "reward_total_mean": 0.856600284576416, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.856600284576416, "rewards/meter/std": 0.34479716420173645, "rewards/total_composite/mean": 0.856600284576416, "rewards/total_composite/std": 0.34479716420173645, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0328437089920044, "sampling/importance_sampling_ratio/min": 0.2593540549278259, "sampling/sampling_logp_difference/max": 1.3495612144470215, "sampling/sampling_logp_difference/mean": 0.18150335550308228, "step": 4 }, { "clip_ratio/high_max": 0.08046106435358524, "clip_ratio/high_mean": 0.08046106435358524, "clip_ratio/low_mean": 0.15030906535685062, "clip_ratio/low_min": 0.15030906535685062, "clip_ratio/region_mean": 0.23077012971043587, "completions/clipped_ratio": 0.0, "completions/max_length": 227.0, "completions/max_terminated_length": 227.0, "completions/mean_length": 163.0, "completions/mean_terminated_length": 163.0, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 2.7085169553756714, "epoch": 0.0001930800123571208, "frac_reward_zero_std": 0.0, "grad_norm": 9.840316772460938, "learning_rate": 9.987878787878788e-06, "loss": -0.0437, "num_tokens": 10053.0, "reward": 0.2891073226928711, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.9642857313156128, "reward_count_adherence_std": 0.06613000482320786, "reward_meter_mean": 0.4093240201473236, "reward_meter_std": 0.3055945336818695, "reward_std": 0.30707788467407227, "reward_total_composite_mean": 0.2891073226928711, "reward_total_composite_std": 0.30707788467407227, "reward_total_mean": 0.2891073226928711, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.9642857313156128, "rewards/count_adherence/std": 0.06613000482320786, "rewards/meter/mean": 0.4093240201473236, "rewards/meter/std": 0.3055945336818695, "rewards/total_composite/mean": 0.2891073226928711, "rewards/total_composite/std": 0.30707788467407227, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0441721677780151, "sampling/importance_sampling_ratio/min": 0.0970935970544815, "sampling/sampling_logp_difference/max": 2.3320798873901367, "sampling/sampling_logp_difference/mean": 0.2513454854488373, "step": 5 }, { "clip_ratio/high_max": 0.11287152394652367, "clip_ratio/high_mean": 0.11287152394652367, "clip_ratio/low_mean": 0.1318096686154604, "clip_ratio/low_min": 0.1318096686154604, "clip_ratio/region_mean": 0.24468119256198406, "completions/clipped_ratio": 0.0, "completions/max_length": 146.0, "completions/max_terminated_length": 146.0, "completions/mean_length": 95.125, "completions/mean_terminated_length": 95.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 2.5477894991636276, "epoch": 0.00023169601482854495, "frac_reward_zero_std": 0.0, "grad_norm": 12.211259841918945, "learning_rate": 9.984848484848485e-06, "loss": -0.0554, "num_tokens": 12078.0, "reward": 0.5588854551315308, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5617634057998657, "reward_meter_std": 0.41573381423950195, "reward_std": 0.42005324363708496, "reward_total_composite_mean": 0.5588854551315308, "reward_total_composite_std": 0.42005324363708496, "reward_total_mean": 0.5588854551315308, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5617634057998657, "rewards/meter/std": 0.41573381423950195, "rewards/total_composite/mean": 0.5588854551315308, "rewards/total_composite/std": 0.42005324363708496, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0142629146575928, "sampling/importance_sampling_ratio/min": 0.17717412114143372, "sampling/sampling_logp_difference/max": 1.7306222915649414, "sampling/sampling_logp_difference/mean": 0.23301345109939575, "step": 6 }, { "clip_ratio/high_max": 0.12872153520584106, "clip_ratio/high_mean": 0.12872153520584106, "clip_ratio/low_mean": 0.10475845448672771, "clip_ratio/low_min": 0.10475845448672771, "clip_ratio/region_mean": 0.23347998969256878, "completions/clipped_ratio": 0.0, "completions/max_length": 146.0, "completions/max_terminated_length": 146.0, "completions/mean_length": 84.875, "completions/mean_terminated_length": 84.875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 2.2808013558387756, "epoch": 0.0002703120172999691, "frac_reward_zero_std": 0.0, "grad_norm": 15.44117546081543, "learning_rate": 9.981818181818183e-06, "loss": 0.2296, "num_tokens": 14125.0, "reward": 0.6451665163040161, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6451665163040161, "reward_meter_std": 0.29868313670158386, "reward_std": 0.29868316650390625, "reward_total_composite_mean": 0.6451665163040161, "reward_total_composite_std": 0.29868313670158386, "reward_total_mean": 0.6451665163040161, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6451665163040161, "rewards/meter/std": 0.29868313670158386, "rewards/total_composite/mean": 0.6451665163040161, "rewards/total_composite/std": 0.29868313670158386, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0209758281707764, "sampling/importance_sampling_ratio/min": 0.1656799614429474, "sampling/sampling_logp_difference/max": 1.7976973056793213, "sampling/sampling_logp_difference/mean": 0.2560131847858429, "step": 7 }, { "clip_ratio/high_max": 0.19223029538989067, "clip_ratio/high_mean": 0.19223029538989067, "clip_ratio/low_mean": 0.03434433601796627, "clip_ratio/low_min": 0.03434433601796627, "clip_ratio/region_mean": 0.22657463140785694, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 56.25, "completions/mean_terminated_length": 56.25, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 1.992170736193657, "epoch": 0.00030892801977139327, "frac_reward_zero_std": 0.0, "grad_norm": 17.118240356445312, "learning_rate": 9.97878787878788e-06, "loss": 0.1614, "num_tokens": 15903.0, "reward": 0.8440761566162109, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8440761566162109, "reward_meter_std": 0.24688278138637543, "reward_std": 0.24688279628753662, "reward_total_composite_mean": 0.8440761566162109, "reward_total_composite_std": 0.24688278138637543, "reward_total_mean": 0.8440761566162109, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8440761566162109, "rewards/meter/std": 0.24688278138637543, "rewards/total_composite/mean": 0.8440761566162109, "rewards/total_composite/std": 0.24688278138637543, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0035207271575928, "sampling/importance_sampling_ratio/min": 0.1740655153989792, "sampling/sampling_logp_difference/max": 1.7483234405517578, "sampling/sampling_logp_difference/mean": 0.22910818457603455, "step": 8 }, { "clip_ratio/high_max": 0.10056564025580883, "clip_ratio/high_mean": 0.10056564025580883, "clip_ratio/low_mean": 0.07851519016548991, "clip_ratio/low_min": 0.07851519016548991, "clip_ratio/region_mean": 0.17908083042129874, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 47.0, "completions/mean_terminated_length": 47.0, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 1.6416602730751038, "epoch": 0.00034754402224281743, "frac_reward_zero_std": 0.0, "grad_norm": 16.50942039489746, "learning_rate": 9.975757575757577e-06, "loss": 0.0706, "num_tokens": 17535.0, "reward": 0.6138224601745605, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6138224601745605, "reward_meter_std": 0.3940174877643585, "reward_std": 0.3940175175666809, "reward_total_composite_mean": 0.6138224601745605, "reward_total_composite_std": 0.3940174877643585, "reward_total_mean": 0.6138224601745605, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6138224601745605, "rewards/meter/std": 0.3940174877643585, "rewards/total_composite/mean": 0.6138224601745605, "rewards/total_composite/std": 0.3940174877643585, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0077967643737793, "sampling/importance_sampling_ratio/min": 0.12568823993206024, "sampling/sampling_logp_difference/max": 2.07395076751709, "sampling/sampling_logp_difference/mean": 0.20549029111862183, "step": 9 }, { "clip_ratio/high_max": 0.1666986495256424, "clip_ratio/high_mean": 0.1666986495256424, "clip_ratio/low_mean": 0.05835868790745735, "clip_ratio/low_min": 0.05835868790745735, "clip_ratio/region_mean": 0.22505733743309975, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 56.5, "completions/mean_terminated_length": 56.5, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 2.4850480556488037, "epoch": 0.0003861600247142416, "frac_reward_zero_std": 0.0, "grad_norm": 18.21146583557129, "learning_rate": 9.972727272727274e-06, "loss": 0.2066, "num_tokens": 19339.0, "reward": 0.7867398858070374, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.8475043773651123, "reward_meter_std": 0.340963751077652, "reward_std": 0.35842904448509216, "reward_total_composite_mean": 0.7867398858070374, "reward_total_composite_std": 0.35842904448509216, "reward_total_mean": 0.7867398858070374, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.8475043773651123, "rewards/meter/std": 0.340963751077652, "rewards/total_composite/mean": 0.7867398858070374, "rewards/total_composite/std": 0.35842904448509216, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.043230414390564, "sampling/importance_sampling_ratio/min": 0.30306050181388855, "sampling/sampling_logp_difference/max": 1.291860580444336, "sampling/sampling_logp_difference/mean": 0.22162646055221558, "step": 10 }, { "clip_ratio/high_max": 0.12178206816315651, "clip_ratio/high_mean": 0.12178206816315651, "clip_ratio/low_mean": 0.09850872680544853, "clip_ratio/low_min": 0.09850872680544853, "clip_ratio/region_mean": 0.22029079496860504, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 44.125, "completions/mean_terminated_length": 44.125, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 1.80090893805027, "epoch": 0.00042477602718566575, "frac_reward_zero_std": 0.0, "grad_norm": 18.612279891967773, "learning_rate": 9.96969696969697e-06, "loss": -0.0602, "num_tokens": 21060.0, "reward": 0.3987988829612732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3987988829612732, "reward_meter_std": 0.39860498905181885, "reward_std": 0.39860498905181885, "reward_total_composite_mean": 0.3987988829612732, "reward_total_composite_std": 0.39860498905181885, "reward_total_mean": 0.3987988829612732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3987988829612732, "rewards/meter/std": 0.39860498905181885, "rewards/total_composite/mean": 0.3987988829612732, "rewards/total_composite/std": 0.39860498905181885, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0170056819915771, "sampling/importance_sampling_ratio/min": 0.15322603285312653, "sampling/sampling_logp_difference/max": 1.8758411407470703, "sampling/sampling_logp_difference/mean": 0.2347755879163742, "step": 11 }, { "clip_ratio/high_max": 0.05123806092888117, "clip_ratio/high_mean": 0.05123806092888117, "clip_ratio/low_mean": 0.12340508960187435, "clip_ratio/low_min": 0.12340508960187435, "clip_ratio/region_mean": 0.17464315053075552, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.9041207674890757, "epoch": 0.0004633920296570899, "frac_reward_zero_std": 0.0, "grad_norm": 18.964420318603516, "learning_rate": 9.966666666666667e-06, "loss": 0.0342, "num_tokens": 23077.0, "reward": 0.5058028697967529, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5058028697967529, "reward_meter_std": 0.40144890546798706, "reward_std": 0.40144890546798706, "reward_total_composite_mean": 0.5058028697967529, "reward_total_composite_std": 0.40144890546798706, "reward_total_mean": 0.5058028697967529, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5058028697967529, "rewards/meter/std": 0.40144890546798706, "rewards/total_composite/mean": 0.5058028697967529, "rewards/total_composite/std": 0.40144890546798706, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9842817783355713, "sampling/importance_sampling_ratio/min": 0.056828152388334274, "sampling/sampling_logp_difference/max": 2.8677234649658203, "sampling/sampling_logp_difference/mean": 0.21280689537525177, "step": 12 }, { "clip_ratio/high_max": 0.1320821400731802, "clip_ratio/high_mean": 0.1320821400731802, "clip_ratio/low_mean": 0.08375322818756104, "clip_ratio/low_min": 0.08375322818756104, "clip_ratio/region_mean": 0.21583536826074123, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 57.875, "completions/mean_terminated_length": 57.875, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 2.2102586179971695, "epoch": 0.000502008032128514, "frac_reward_zero_std": 0.0, "grad_norm": 15.997991561889648, "learning_rate": 9.963636363636364e-06, "loss": 0.2512, "num_tokens": 24884.0, "reward": 0.45139390230178833, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.45139390230178833, "reward_meter_std": 0.43168777227401733, "reward_std": 0.4316878020763397, "reward_total_composite_mean": 0.45139390230178833, "reward_total_composite_std": 0.43168777227401733, "reward_total_mean": 0.45139390230178833, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.45139390230178833, "rewards/meter/std": 0.43168777227401733, "rewards/total_composite/mean": 0.45139390230178833, "rewards/total_composite/std": 0.43168777227401733, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0242526531219482, "sampling/importance_sampling_ratio/min": 0.25717854499816895, "sampling/sampling_logp_difference/max": 1.3626043796539307, "sampling/sampling_logp_difference/mean": 0.2293073683977127, "step": 13 }, { "clip_ratio/high_max": 0.10542293824255466, "clip_ratio/high_mean": 0.10542293824255466, "clip_ratio/low_mean": 0.09186305850744247, "clip_ratio/low_min": 0.09186305850744247, "clip_ratio/region_mean": 0.19728599674999714, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 107.375, "completions/mean_terminated_length": 107.375, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 1.8395927399396896, "epoch": 0.0005406240345999382, "frac_reward_zero_std": 0.0, "grad_norm": 12.638867378234863, "learning_rate": 9.960606060606062e-06, "loss": -0.029, "num_tokens": 27287.0, "reward": 0.5478648543357849, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5478648543357849, "reward_meter_std": 0.35016050934791565, "reward_std": 0.35016050934791565, "reward_total_composite_mean": 0.5478648543357849, "reward_total_composite_std": 0.35016050934791565, "reward_total_mean": 0.5478648543357849, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5478648543357849, "rewards/meter/std": 0.35016050934791565, "rewards/total_composite/mean": 0.5478648543357849, "rewards/total_composite/std": 0.35016050934791565, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.019142508506775, "sampling/importance_sampling_ratio/min": 0.15102817118167877, "sampling/sampling_logp_difference/max": 1.8902888298034668, "sampling/sampling_logp_difference/mean": 0.2364599108695984, "step": 14 }, { "clip_ratio/high_max": 0.1045012567192316, "clip_ratio/high_mean": 0.1045012567192316, "clip_ratio/low_mean": 0.08156565949320793, "clip_ratio/low_min": 0.08156565949320793, "clip_ratio/region_mean": 0.18606691621243954, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 25.125, "completions/mean_terminated_length": 25.125, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 1.8547815531492233, "epoch": 0.0005792400370713623, "frac_reward_zero_std": 0.0, "grad_norm": 23.416534423828125, "learning_rate": 9.957575757575757e-06, "loss": 0.0038, "num_tokens": 28664.0, "reward": 0.4547968804836273, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4547968804836273, "reward_meter_std": 0.4638686776161194, "reward_std": 0.463868647813797, "reward_total_composite_mean": 0.4547968804836273, "reward_total_composite_std": 0.4638686776161194, "reward_total_mean": 0.4547968804836273, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4547968804836273, "rewards/meter/std": 0.4638686776161194, "rewards/total_composite/mean": 0.4547968804836273, "rewards/total_composite/std": 0.4638686776161194, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0308705568313599, "sampling/importance_sampling_ratio/min": 0.34662288427352905, "sampling/sampling_logp_difference/max": 1.0595178604125977, "sampling/sampling_logp_difference/mean": 0.17740340530872345, "step": 15 }, { "clip_ratio/high_max": 0.17772199772298336, "clip_ratio/high_mean": 0.17772199772298336, "clip_ratio/low_mean": 0.02777777798473835, "clip_ratio/low_min": 0.02777777798473835, "clip_ratio/region_mean": 0.2054997757077217, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 42.125, "completions/mean_terminated_length": 42.125, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 1.8997105956077576, "epoch": 0.0006178560395427865, "frac_reward_zero_std": 0.0, "grad_norm": 17.291547775268555, "learning_rate": 9.954545454545456e-06, "loss": 0.1985, "num_tokens": 30313.0, "reward": 0.8227195143699646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8227195143699646, "reward_meter_std": 0.30595460534095764, "reward_std": 0.30595454573631287, "reward_total_composite_mean": 0.8227195143699646, "reward_total_composite_std": 0.30595460534095764, "reward_total_mean": 0.8227195143699646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8227195143699646, "rewards/meter/std": 0.30595460534095764, "rewards/total_composite/mean": 0.8227195143699646, "rewards/total_composite/std": 0.30595460534095764, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0267019271850586, "sampling/importance_sampling_ratio/min": 0.2083941400051117, "sampling/sampling_logp_difference/max": 1.568324089050293, "sampling/sampling_logp_difference/mean": 0.2161049097776413, "step": 16 }, { "clip_ratio/high_max": 0.02809617994353175, "clip_ratio/high_mean": 0.02809617994353175, "clip_ratio/low_mean": 0.05947580561041832, "clip_ratio/low_min": 0.05947580561041832, "clip_ratio/region_mean": 0.08757198555395007, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 44.375, "completions/mean_terminated_length": 44.375, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.43756548687815666, "epoch": 0.0006564720420142106, "frac_reward_zero_std": 0.0, "grad_norm": 14.35202693939209, "learning_rate": 9.951515151515152e-06, "loss": -0.0398, "num_tokens": 31900.0, "reward": 0.8780908584594727, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8780908584594727, "reward_meter_std": 0.1825910061597824, "reward_std": 0.1825910061597824, "reward_total_composite_mean": 0.8780908584594727, "reward_total_composite_std": 0.1825910061597824, "reward_total_mean": 0.8780908584594727, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8780908584594727, "rewards/meter/std": 0.1825910061597824, "rewards/total_composite/mean": 0.8780908584594727, "rewards/total_composite/std": 0.1825910061597824, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9951265454292297, "sampling/importance_sampling_ratio/min": 0.12622620165348053, "sampling/sampling_logp_difference/max": 2.0696797370910645, "sampling/sampling_logp_difference/mean": 0.09773284941911697, "step": 17 }, { "clip_ratio/high_max": 0.07820177916437387, "clip_ratio/high_mean": 0.07820177916437387, "clip_ratio/low_mean": 0.14138433896005154, "clip_ratio/low_min": 0.14138433896005154, "clip_ratio/region_mean": 0.2195861181244254, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 46.75, "completions/mean_terminated_length": 46.75, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 2.013083055615425, "epoch": 0.0006950880444856349, "frac_reward_zero_std": 0.0, "grad_norm": 17.070714950561523, "learning_rate": 9.948484848484849e-06, "loss": 0.2567, "num_tokens": 33490.0, "reward": 0.45992743968963623, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.45992743968963623, "reward_meter_std": 0.3919222354888916, "reward_std": 0.3919222354888916, "reward_total_composite_mean": 0.45992743968963623, "reward_total_composite_std": 0.3919222354888916, "reward_total_mean": 0.45992743968963623, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.45992743968963623, "rewards/meter/std": 0.3919222354888916, "rewards/total_composite/mean": 0.45992743968963623, "rewards/total_composite/std": 0.3919222354888916, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0427944660186768, "sampling/importance_sampling_ratio/min": 0.23565533757209778, "sampling/sampling_logp_difference/max": 1.4453849792480469, "sampling/sampling_logp_difference/mean": 0.21356253325939178, "step": 18 }, { "clip_ratio/high_max": 0.05499837175011635, "clip_ratio/high_mean": 0.05499837175011635, "clip_ratio/low_mean": 0.1405880395323038, "clip_ratio/low_min": 0.1405880395323038, "clip_ratio/region_mean": 0.19558641128242016, "completions/clipped_ratio": 0.0, "completions/max_length": 151.0, "completions/max_terminated_length": 151.0, "completions/mean_length": 118.25, "completions/mean_terminated_length": 118.25, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 2.113861307501793, "epoch": 0.000733704046957059, "frac_reward_zero_std": 0.0, "grad_norm": 8.703368186950684, "learning_rate": 9.945454545454546e-06, "loss": 0.045, "num_tokens": 35844.0, "reward": 0.22483205795288086, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2513314485549927, "reward_meter_std": 0.1909521371126175, "reward_std": 0.2108626812696457, "reward_total_composite_mean": 0.22483205795288086, "reward_total_composite_std": 0.2108626663684845, "reward_total_mean": 0.22483205795288086, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2513314485549927, "rewards/meter/std": 0.1909521371126175, "rewards/total_composite/mean": 0.22483205795288086, "rewards/total_composite/std": 0.2108626663684845, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0284098386764526, "sampling/importance_sampling_ratio/min": 0.1471317857503891, "sampling/sampling_logp_difference/max": 1.916426658630371, "sampling/sampling_logp_difference/mean": 0.20431792736053467, "step": 19 }, { "clip_ratio/high_max": 0.11636904999613762, "clip_ratio/high_mean": 0.11636904999613762, "clip_ratio/low_mean": 0.09608769603073597, "clip_ratio/low_min": 0.09608769603073597, "clip_ratio/region_mean": 0.2124567460268736, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 59.375, "completions/mean_terminated_length": 59.375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 2.4152235835790634, "epoch": 0.0007723200494284832, "frac_reward_zero_std": 0.0, "grad_norm": 14.549918174743652, "learning_rate": 9.942424242424244e-06, "loss": 0.0584, "num_tokens": 37711.0, "reward": 0.5284043550491333, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.5680004358291626, "reward_meter_std": 0.3963819742202759, "reward_std": 0.44208046793937683, "reward_total_composite_mean": 0.5284043550491333, "reward_total_composite_std": 0.44208046793937683, "reward_total_mean": 0.5284043550491333, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.5680004358291626, "rewards/meter/std": 0.3963819742202759, "rewards/total_composite/mean": 0.5284043550491333, "rewards/total_composite/std": 0.44208046793937683, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0309125185012817, "sampling/importance_sampling_ratio/min": 0.14759160578250885, "sampling/sampling_logp_difference/max": 1.9133062362670898, "sampling/sampling_logp_difference/mean": 0.24648143351078033, "step": 20 }, { "clip_ratio/high_max": 0.07960096746683121, "clip_ratio/high_mean": 0.07960096746683121, "clip_ratio/low_mean": 0.10977440886199474, "clip_ratio/low_min": 0.10977440886199474, "clip_ratio/region_mean": 0.18937537632882595, "completions/clipped_ratio": 0.0, "completions/max_length": 195.0, "completions/max_terminated_length": 195.0, "completions/mean_length": 171.625, "completions/mean_terminated_length": 171.625, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 2.130246579647064, "epoch": 0.0008109360518999073, "frac_reward_zero_std": 0.0, "grad_norm": 6.904236316680908, "learning_rate": 9.939393939393939e-06, "loss": 0.0205, "num_tokens": 40580.0, "reward": 0.22051820158958435, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.29830992221832275, "reward_meter_std": 0.28082314133644104, "reward_std": 0.3221690356731415, "reward_total_composite_mean": 0.22051820158958435, "reward_total_composite_std": 0.3221690356731415, "reward_total_mean": 0.22051820158958435, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.29830992221832275, "rewards/meter/std": 0.28082314133644104, "rewards/total_composite/mean": 0.22051820158958435, "rewards/total_composite/std": 0.3221690356731415, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0334407091140747, "sampling/importance_sampling_ratio/min": 0.24302920699119568, "sampling/sampling_logp_difference/max": 1.4145736694335938, "sampling/sampling_logp_difference/mean": 0.18756546080112457, "step": 21 }, { "clip_ratio/high_max": 0.07555159274488688, "clip_ratio/high_mean": 0.07555159274488688, "clip_ratio/low_mean": 0.12670023273676634, "clip_ratio/low_min": 0.12670023273676634, "clip_ratio/region_mean": 0.2022518254816532, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 46.5, "completions/mean_terminated_length": 46.5, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 1.631166860461235, "epoch": 0.0008495520543713315, "frac_reward_zero_std": 0.0, "grad_norm": 15.95211124420166, "learning_rate": 9.936363636363638e-06, "loss": -0.151, "num_tokens": 42240.0, "reward": 0.48668479919433594, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.48668479919433594, "reward_meter_std": 0.44610869884490967, "reward_std": 0.4461086690425873, "reward_total_composite_mean": 0.48668479919433594, "reward_total_composite_std": 0.44610869884490967, "reward_total_mean": 0.48668479919433594, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.48668479919433594, "rewards/meter/std": 0.44610869884490967, "rewards/total_composite/mean": 0.48668479919433594, "rewards/total_composite/std": 0.44610869884490967, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.032152533531189, "sampling/importance_sampling_ratio/min": 0.13872095942497253, "sampling/sampling_logp_difference/max": 1.9752907752990723, "sampling/sampling_logp_difference/mean": 0.19472216069698334, "step": 22 }, { "clip_ratio/high_max": 0.12303969636559486, "clip_ratio/high_mean": 0.12303969636559486, "clip_ratio/low_mean": 0.09274027217179537, "clip_ratio/low_min": 0.09274027217179537, "clip_ratio/region_mean": 0.21577996853739023, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 59.875, "completions/mean_terminated_length": 59.875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 2.0983588993549347, "epoch": 0.0008881680568427556, "frac_reward_zero_std": 0.0, "grad_norm": 12.759511947631836, "learning_rate": 9.933333333333334e-06, "loss": 0.0144, "num_tokens": 44023.0, "reward": 0.5005471706390381, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5005471706390381, "reward_meter_std": 0.4535389840602875, "reward_std": 0.4535389840602875, "reward_total_composite_mean": 0.5005471706390381, "reward_total_composite_std": 0.4535389840602875, "reward_total_mean": 0.5005471706390381, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5005471706390381, "rewards/meter/std": 0.4535389840602875, "rewards/total_composite/mean": 0.5005471706390381, "rewards/total_composite/std": 0.4535389840602875, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0337328910827637, "sampling/importance_sampling_ratio/min": 0.2599697709083557, "sampling/sampling_logp_difference/max": 1.3471899032592773, "sampling/sampling_logp_difference/mean": 0.20020200312137604, "step": 23 }, { "clip_ratio/high_max": 0.11741364374756813, "clip_ratio/high_mean": 0.11741364374756813, "clip_ratio/low_mean": 0.09042087476700544, "clip_ratio/low_min": 0.09042087476700544, "clip_ratio/region_mean": 0.20783451851457357, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 50.625, "completions/mean_terminated_length": 50.625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 2.2059893161058426, "epoch": 0.0009267840593141798, "frac_reward_zero_std": 0.0, "grad_norm": 18.0654354095459, "learning_rate": 9.930303030303031e-06, "loss": 0.0997, "num_tokens": 45780.0, "reward": 0.4580531120300293, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.6220148801803589, "reward_meter_std": 0.4630615711212158, "reward_std": 0.46495571732521057, "reward_total_composite_mean": 0.4580531120300293, "reward_total_composite_std": 0.46495571732521057, "reward_total_mean": 0.4580531120300293, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.6220148801803589, "rewards/meter/std": 0.4630615711212158, "rewards/total_composite/mean": 0.4580531120300293, "rewards/total_composite/std": 0.46495571732521057, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0300155878067017, "sampling/importance_sampling_ratio/min": 0.22524529695510864, "sampling/sampling_logp_difference/max": 1.490565299987793, "sampling/sampling_logp_difference/mean": 0.2604425549507141, "step": 24 }, { "clip_ratio/high_max": 0.08282828330993652, "clip_ratio/high_mean": 0.08282828330993652, "clip_ratio/low_mean": 0.09993548225611448, "clip_ratio/low_min": 0.09993548225611448, "clip_ratio/region_mean": 0.182763765566051, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 44.375, "completions/mean_terminated_length": 44.375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 1.3184484615921974, "epoch": 0.0009654000617856039, "frac_reward_zero_std": 0.0, "grad_norm": 15.530243873596191, "learning_rate": 9.927272727272728e-06, "loss": 0.0988, "num_tokens": 47479.0, "reward": 0.3527180254459381, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3527180254459381, "reward_meter_std": 0.4116312861442566, "reward_std": 0.4116312563419342, "reward_total_composite_mean": 0.3527180254459381, "reward_total_composite_std": 0.4116312861442566, "reward_total_mean": 0.3527180254459381, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3527180254459381, "rewards/meter/std": 0.4116312861442566, "rewards/total_composite/mean": 0.3527180254459381, "rewards/total_composite/std": 0.4116312861442566, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.016542911529541, "sampling/importance_sampling_ratio/min": 0.20795388519763947, "sampling/sampling_logp_difference/max": 1.5704388618469238, "sampling/sampling_logp_difference/mean": 0.17016799747943878, "step": 25 }, { "clip_ratio/high_max": 0.17995928972959518, "clip_ratio/high_mean": 0.17995928972959518, "clip_ratio/low_mean": 0.02651515230536461, "clip_ratio/low_min": 0.02651515230536461, "clip_ratio/region_mean": 0.2064744420349598, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 44.5, "completions/mean_terminated_length": 44.5, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 1.9691928625106812, "epoch": 0.001004016064257028, "frac_reward_zero_std": 0.0, "grad_norm": 21.890466690063477, "learning_rate": 9.924242424242425e-06, "loss": -0.082, "num_tokens": 49115.0, "reward": 0.8331155776977539, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8464770913124084, "reward_meter_std": 0.3041004538536072, "reward_std": 0.3413103222846985, "reward_total_composite_mean": 0.8331155776977539, "reward_total_composite_std": 0.3413103222846985, "reward_total_mean": 0.8331155776977539, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8464770913124084, "rewards/meter/std": 0.3041004538536072, "rewards/total_composite/mean": 0.8331155776977539, "rewards/total_composite/std": 0.3413103222846985, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0382486581802368, "sampling/importance_sampling_ratio/min": 0.14871208369731903, "sampling/sampling_logp_difference/max": 1.90574312210083, "sampling/sampling_logp_difference/mean": 0.25624310970306396, "step": 26 }, { "clip_ratio/high_max": 0.15723814070224762, "clip_ratio/high_mean": 0.15723814070224762, "clip_ratio/low_mean": 0.04641487076878548, "clip_ratio/low_min": 0.04641487076878548, "clip_ratio/region_mean": 0.2036530114710331, "completions/clipped_ratio": 0.0, "completions/max_length": 176.0, "completions/max_terminated_length": 176.0, "completions/mean_length": 159.875, "completions/mean_terminated_length": 159.875, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "entropy": 1.940888598561287, "epoch": 0.0010426320667284523, "frac_reward_zero_std": 0.0, "grad_norm": 8.763998031616211, "learning_rate": 9.921212121212121e-06, "loss": 0.0432, "num_tokens": 51914.0, "reward": 0.44596123695373535, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.44596123695373535, "reward_meter_std": 0.3097141981124878, "reward_std": 0.3097141981124878, "reward_total_composite_mean": 0.44596123695373535, "reward_total_composite_std": 0.3097141981124878, "reward_total_mean": 0.44596123695373535, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.44596123695373535, "rewards/meter/std": 0.3097141981124878, "rewards/total_composite/mean": 0.44596123695373535, "rewards/total_composite/std": 0.3097141981124878, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0340567827224731, "sampling/importance_sampling_ratio/min": 0.16828162968158722, "sampling/sampling_logp_difference/max": 1.782116413116455, "sampling/sampling_logp_difference/mean": 0.19997085630893707, "step": 27 }, { "clip_ratio/high_max": 0.08741745911538601, "clip_ratio/high_mean": 0.08741745911538601, "clip_ratio/low_mean": 0.10222199466079473, "clip_ratio/low_min": 0.10222199466079473, "clip_ratio/region_mean": 0.18963945377618074, "completions/clipped_ratio": 0.0, "completions/max_length": 223.0, "completions/max_terminated_length": 223.0, "completions/mean_length": 122.125, "completions/mean_terminated_length": 122.125, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 1.0821578055620193, "epoch": 0.0010812480691998764, "frac_reward_zero_std": 0.0, "grad_norm": 16.314180374145508, "learning_rate": 9.918181818181818e-06, "loss": -0.0001, "num_tokens": 54507.0, "reward": 0.06344515830278397, "reward_arabic_clean_mean": 0.375, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.4260466396808624, "reward_meter_std": 0.2623087763786316, "reward_std": 0.09849678725004196, "reward_total_composite_mean": 0.06344515830278397, "reward_total_composite_std": 0.09849678725004196, "reward_total_mean": 0.06344515830278397, "rewards/arabic_clean/mean": 0.375, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.4260466396808624, "rewards/meter/std": 0.2623087763786316, "rewards/total_composite/mean": 0.06344515830278397, "rewards/total_composite/std": 0.09849678725004196, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9840916395187378, "sampling/importance_sampling_ratio/min": 0.03912108764052391, "sampling/sampling_logp_difference/max": 3.241093635559082, "sampling/sampling_logp_difference/mean": 0.2858719229698181, "step": 28 }, { "clip_ratio/high_max": 0.17736977525055408, "clip_ratio/high_mean": 0.17736977525055408, "clip_ratio/low_mean": 0.05003259517252445, "clip_ratio/low_min": 0.05003259517252445, "clip_ratio/region_mean": 0.22740237042307854, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 44.75, "completions/mean_terminated_length": 44.75, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 1.6001797169446945, "epoch": 0.0011198640716713006, "frac_reward_zero_std": 0.0, "grad_norm": 18.285869598388672, "learning_rate": 9.915151515151515e-06, "loss": 0.0883, "num_tokens": 56313.0, "reward": 0.7853162884712219, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7853162884712219, "reward_meter_std": 0.29714658856391907, "reward_std": 0.29714658856391907, "reward_total_composite_mean": 0.7853162884712219, "reward_total_composite_std": 0.29714658856391907, "reward_total_mean": 0.7853162884712219, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7853162884712219, "rewards/meter/std": 0.29714658856391907, "rewards/total_composite/mean": 0.7853162884712219, "rewards/total_composite/std": 0.29714658856391907, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9876272082328796, "sampling/importance_sampling_ratio/min": 0.10404817014932632, "sampling/sampling_logp_difference/max": 2.2629013061523438, "sampling/sampling_logp_difference/mean": 0.20776626467704773, "step": 29 }, { "clip_ratio/high_max": 0.1195354238152504, "clip_ratio/high_mean": 0.1195354238152504, "clip_ratio/low_mean": 0.09173617325723171, "clip_ratio/low_min": 0.09173617325723171, "clip_ratio/region_mean": 0.2112715970724821, "completions/clipped_ratio": 0.0, "completions/max_length": 220.0, "completions/max_terminated_length": 220.0, "completions/mean_length": 169.625, "completions/mean_terminated_length": 169.625, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 2.6212728768587112, "epoch": 0.0011584800741427247, "frac_reward_zero_std": 0.0, "grad_norm": 8.281560897827148, "learning_rate": 9.912121212121213e-06, "loss": 0.0412, "num_tokens": 59278.0, "reward": 0.48327285051345825, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5359864234924316, "reward_meter_std": 0.21292957663536072, "reward_std": 0.2851979732513428, "reward_total_composite_mean": 0.48327285051345825, "reward_total_composite_std": 0.28519800305366516, "reward_total_mean": 0.48327285051345825, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5359864234924316, "rewards/meter/std": 0.21292957663536072, "rewards/total_composite/mean": 0.48327285051345825, "rewards/total_composite/std": 0.28519800305366516, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0370231866836548, "sampling/importance_sampling_ratio/min": 0.1704005002975464, "sampling/sampling_logp_difference/max": 1.7696037292480469, "sampling/sampling_logp_difference/mean": 0.227935329079628, "step": 30 }, { "clip_ratio/high_max": 0.16296951659023762, "clip_ratio/high_mean": 0.16296951659023762, "clip_ratio/low_mean": 0.06649214401841164, "clip_ratio/low_min": 0.06649214401841164, "clip_ratio/region_mean": 0.22946166060864925, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 67.875, "completions/mean_terminated_length": 67.875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 2.164387509226799, "epoch": 0.001197096076614149, "frac_reward_zero_std": 0.0, "grad_norm": 14.505411148071289, "learning_rate": 9.90909090909091e-06, "loss": 0.0224, "num_tokens": 61117.0, "reward": 0.6783833503723145, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6783833503723145, "reward_meter_std": 0.4296880066394806, "reward_std": 0.4296879768371582, "reward_total_composite_mean": 0.6783833503723145, "reward_total_composite_std": 0.4296880066394806, "reward_total_mean": 0.6783833503723145, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6783833503723145, "rewards/meter/std": 0.4296880066394806, "rewards/total_composite/mean": 0.6783833503723145, "rewards/total_composite/std": 0.4296880066394806, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0112130641937256, "sampling/importance_sampling_ratio/min": 0.2134408801794052, "sampling/sampling_logp_difference/max": 1.5443954467773438, "sampling/sampling_logp_difference/mean": 0.24672164022922516, "step": 31 }, { "clip_ratio/high_max": 0.02909391513094306, "clip_ratio/high_mean": 0.02909391513094306, "clip_ratio/low_mean": 0.03902714978903532, "clip_ratio/low_min": 0.03902714978903532, "clip_ratio/region_mean": 0.06812106491997838, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 27.625, "completions/mean_terminated_length": 27.625, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 0.44207035191357136, "epoch": 0.001235712079085573, "frac_reward_zero_std": 0.0, "grad_norm": 22.633512496948242, "learning_rate": 9.906060606060607e-06, "loss": 0.0377, "num_tokens": 62698.0, "reward": 0.9473069310188293, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9473069310188293, "reward_meter_std": 0.03623950853943825, "reward_std": 0.03623950108885765, "reward_total_composite_mean": 0.9473069310188293, "reward_total_composite_std": 0.03623950853943825, "reward_total_mean": 0.9473069310188293, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9473069310188293, "rewards/meter/std": 0.03623950853943825, "rewards/total_composite/mean": 0.9473069310188293, "rewards/total_composite/std": 0.03623950853943825, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9921619892120361, "sampling/importance_sampling_ratio/min": 0.0969339981675148, "sampling/sampling_logp_difference/max": 2.3337249755859375, "sampling/sampling_logp_difference/mean": 0.09311103075742722, "step": 32 }, { "clip_ratio/high_max": 0.08508810587227345, "clip_ratio/high_mean": 0.08508810587227345, "clip_ratio/low_mean": 0.14664593152701855, "clip_ratio/low_min": 0.14664593152701855, "clip_ratio/region_mean": 0.231734037399292, "completions/clipped_ratio": 0.0, "completions/max_length": 274.0, "completions/max_terminated_length": 274.0, "completions/mean_length": 181.125, "completions/mean_terminated_length": 181.125, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 2.361170306801796, "epoch": 0.0012743280815569972, "frac_reward_zero_std": 0.0, "grad_norm": 9.995526313781738, "learning_rate": 9.903030303030305e-06, "loss": -0.1309, "num_tokens": 65811.0, "reward": 0.19121383130550385, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.05750546231865883, "reward_meter_mean": 0.2281605750322342, "reward_meter_std": 0.13796716928482056, "reward_std": 0.15800805389881134, "reward_total_composite_mean": 0.19121383130550385, "reward_total_composite_std": 0.15800805389881134, "reward_total_mean": 0.19121383130550385, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.05750546231865883, "rewards/meter/mean": 0.2281605750322342, "rewards/meter/std": 0.13796716928482056, "rewards/total_composite/mean": 0.19121383130550385, "rewards/total_composite/std": 0.15800805389881134, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0408287048339844, "sampling/importance_sampling_ratio/min": 0.10168811678886414, "sampling/sampling_logp_difference/max": 2.2858448028564453, "sampling/sampling_logp_difference/mean": 0.26636290550231934, "step": 33 }, { "clip_ratio/high_max": 0.03525641094893217, "clip_ratio/high_mean": 0.03525641094893217, "clip_ratio/low_mean": 0.09407370304688811, "clip_ratio/low_min": 0.09407370304688811, "clip_ratio/region_mean": 0.12933011399582028, "completions/clipped_ratio": 0.0, "completions/max_length": 31.0, "completions/max_terminated_length": 31.0, "completions/mean_length": 26.375, "completions/mean_terminated_length": 26.375, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.5113434419035912, "epoch": 0.0013129440840284213, "frac_reward_zero_std": 0.0, "grad_norm": 21.821182250976562, "learning_rate": 9.9e-06, "loss": 0.0783, "num_tokens": 67182.0, "reward": 0.1796203851699829, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.1796203851699829, "reward_meter_std": 0.330172061920166, "reward_std": 0.330172061920166, "reward_total_composite_mean": 0.1796203851699829, "reward_total_composite_std": 0.330172061920166, "reward_total_mean": 0.1796203851699829, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.1796203851699829, "rewards/meter/std": 0.330172061920166, "rewards/total_composite/mean": 0.1796203851699829, "rewards/total_composite/std": 0.330172061920166, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0188137292861938, "sampling/importance_sampling_ratio/min": 0.12298180162906647, "sampling/sampling_logp_difference/max": 2.0957188606262207, "sampling/sampling_logp_difference/mean": 0.18016308546066284, "step": 34 }, { "clip_ratio/high_max": 0.07765224599279463, "clip_ratio/high_mean": 0.07765224599279463, "clip_ratio/low_mean": 0.05922202859073877, "clip_ratio/low_min": 0.05922202859073877, "clip_ratio/region_mean": 0.1368742745835334, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 45.875, "completions/mean_terminated_length": 45.875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 1.3972373455762863, "epoch": 0.0013515600864998456, "frac_reward_zero_std": 0.0, "grad_norm": 14.014320373535156, "learning_rate": 9.896969696969699e-06, "loss": -0.0922, "num_tokens": 68837.0, "reward": 0.5994056463241577, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5994056463241577, "reward_meter_std": 0.40437188744544983, "reward_std": 0.40437188744544983, "reward_total_composite_mean": 0.5994056463241577, "reward_total_composite_std": 0.40437188744544983, "reward_total_mean": 0.5994056463241577, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5994056463241577, "rewards/meter/std": 0.40437188744544983, "rewards/total_composite/mean": 0.5994056463241577, "rewards/total_composite/std": 0.40437188744544983, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0345124006271362, "sampling/importance_sampling_ratio/min": 0.31121668219566345, "sampling/sampling_logp_difference/max": 1.3427734375, "sampling/sampling_logp_difference/mean": 0.1685187667608261, "step": 35 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.02442893385887146, "clip_ratio/low_min": 0.02442893385887146, "clip_ratio/region_mean": 0.02442893385887146, "completions/clipped_ratio": 0.875, "completions/max_length": 512.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 497.25, "completions/mean_terminated_length": 394.0, "completions/min_length": 394.0, "completions/min_terminated_length": 394.0, "entropy": 0.48652103543281555, "epoch": 0.0013901760889712697, "frac_reward_zero_std": 0.0, "grad_norm": 3.0867953300476074, "learning_rate": 9.893939393939395e-06, "loss": 0.2225, "num_tokens": 70919.0, "reward": 0.6741494536399841, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9264706373214722, "reward_count_adherence_std": 0.08753220736980438, "reward_meter_mean": 0.86127769947052, "reward_meter_std": 0.20221967995166779, "reward_std": 0.3219583332538605, "reward_total_composite_mean": 0.6741494536399841, "reward_total_composite_std": 0.3219583332538605, "reward_total_mean": 0.6741494536399841, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9264706373214722, "rewards/count_adherence/std": 0.08753220736980438, "rewards/meter/mean": 0.86127769947052, "rewards/meter/std": 0.20221967995166779, "rewards/total_composite/mean": 0.6741494536399841, "rewards/total_composite/std": 0.3219583332538605, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0490732192993164, "sampling/importance_sampling_ratio/min": 0.2985435426235199, "sampling/sampling_logp_difference/max": 1.2088394165039062, "sampling/sampling_logp_difference/mean": 0.2478996217250824, "step": 36 }, { "clip_ratio/high_max": 0.1309125702828169, "clip_ratio/high_mean": 0.1309125702828169, "clip_ratio/low_mean": 0.09653645940124989, "clip_ratio/low_min": 0.09653645940124989, "clip_ratio/region_mean": 0.22744902968406677, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 2.081456035375595, "epoch": 0.0014287920914426938, "frac_reward_zero_std": 0.0, "grad_norm": 15.039297103881836, "learning_rate": 9.890909090909092e-06, "loss": 0.0523, "num_tokens": 72813.0, "reward": 0.5642927289009094, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_meter_mean": 0.6241669654846191, "reward_meter_std": 0.3833092451095581, "reward_std": 0.3661707043647766, "reward_total_composite_mean": 0.5642927289009094, "reward_total_composite_std": 0.3661707043647766, "reward_total_mean": 0.5642927289009094, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/meter/mean": 0.6241669654846191, "rewards/meter/std": 0.3833092451095581, "rewards/total_composite/mean": 0.5642927289009094, "rewards/total_composite/std": 0.3661707043647766, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.026777744293213, "sampling/importance_sampling_ratio/min": 0.2108944207429886, "sampling/sampling_logp_difference/max": 1.5563976764678955, "sampling/sampling_logp_difference/mean": 0.2267686128616333, "step": 37 }, { "clip_ratio/high_max": 0.13196256244555116, "clip_ratio/high_mean": 0.13196256244555116, "clip_ratio/low_mean": 0.0223214291036129, "clip_ratio/low_min": 0.0223214291036129, "clip_ratio/region_mean": 0.15428399154916406, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 26.25, "completions/mean_terminated_length": 26.25, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 1.365403652191162, "epoch": 0.001467408093914118, "frac_reward_zero_std": 0.0, "grad_norm": 17.81291389465332, "learning_rate": 9.887878787878789e-06, "loss": 0.0441, "num_tokens": 74327.0, "reward": 0.9193331003189087, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9193331003189087, "reward_meter_std": 0.18450921773910522, "reward_std": 0.18450923264026642, "reward_total_composite_mean": 0.9193331003189087, "reward_total_composite_std": 0.18450921773910522, "reward_total_mean": 0.9193331003189087, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9193331003189087, "rewards/meter/std": 0.18450921773910522, "rewards/total_composite/mean": 0.9193331003189087, "rewards/total_composite/std": 0.18450921773910522, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9957528710365295, "sampling/importance_sampling_ratio/min": 0.07545739412307739, "sampling/sampling_logp_difference/max": 2.5841870307922363, "sampling/sampling_logp_difference/mean": 0.18091736733913422, "step": 38 }, { "clip_ratio/high_max": 0.03838060609996319, "clip_ratio/high_mean": 0.03838060609996319, "clip_ratio/low_mean": 0.10097905434668064, "clip_ratio/low_min": 0.10097905434668064, "clip_ratio/region_mean": 0.13935966044664383, "completions/clipped_ratio": 0.0, "completions/max_length": 89.0, "completions/max_terminated_length": 89.0, "completions/mean_length": 82.875, "completions/mean_terminated_length": 82.875, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.739346232265234, "epoch": 0.0015060240963855422, "frac_reward_zero_std": 0.0, "grad_norm": 11.80769157409668, "learning_rate": 9.884848484848486e-06, "loss": 0.0578, "num_tokens": 76302.0, "reward": 0.2587703466415405, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2643655240535736, "reward_meter_std": 0.3247142732143402, "reward_std": 0.3293907940387726, "reward_total_composite_mean": 0.2587703466415405, "reward_total_composite_std": 0.32939082384109497, "reward_total_mean": 0.2587703466415405, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2643655240535736, "rewards/meter/std": 0.3247142732143402, "rewards/total_composite/mean": 0.2587703466415405, "rewards/total_composite/std": 0.32939082384109497, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9873656630516052, "sampling/importance_sampling_ratio/min": 1.2175244592071977e-05, "sampling/sampling_logp_difference/max": 11.316105842590332, "sampling/sampling_logp_difference/mean": 0.20199847221374512, "step": 39 }, { "clip_ratio/high_max": 0.18653450906276703, "clip_ratio/high_mean": 0.18653450906276703, "clip_ratio/low_mean": 0.031887754797935486, "clip_ratio/low_min": 0.031887754797935486, "clip_ratio/region_mean": 0.21842226386070251, "completions/clipped_ratio": 0.0, "completions/max_length": 146.0, "completions/max_terminated_length": 146.0, "completions/mean_length": 107.5, "completions/mean_terminated_length": 107.5, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 2.759167045354843, "epoch": 0.0015446400988569664, "frac_reward_zero_std": 0.0, "grad_norm": 9.70837116241455, "learning_rate": 9.881818181818182e-06, "loss": -0.0041, "num_tokens": 78586.0, "reward": 0.8742237687110901, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8742237687110901, "reward_meter_std": 0.31200793385505676, "reward_std": 0.31200793385505676, "reward_total_composite_mean": 0.8742237687110901, "reward_total_composite_std": 0.31200793385505676, "reward_total_mean": 0.8742237687110901, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8742237687110901, "rewards/meter/std": 0.31200793385505676, "rewards/total_composite/mean": 0.8742237687110901, "rewards/total_composite/std": 0.31200793385505676, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.039535641670227, "sampling/importance_sampling_ratio/min": 0.1948380321264267, "sampling/sampling_logp_difference/max": 1.6355867385864258, "sampling/sampling_logp_difference/mean": 0.21297502517700195, "step": 40 }, { "clip_ratio/high_max": 0.09009376727044582, "clip_ratio/high_mean": 0.09009376727044582, "clip_ratio/low_mean": 0.08622571267187595, "clip_ratio/low_min": 0.08622571267187595, "clip_ratio/region_mean": 0.17631947994232178, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 390.875, "completions/mean_terminated_length": 390.875, "completions/min_length": 322.0, "completions/min_terminated_length": 322.0, "entropy": 2.349475011229515, "epoch": 0.0015832561013283905, "frac_reward_zero_std": 0.0, "grad_norm": 5.116574287414551, "learning_rate": 9.87878787878788e-06, "loss": 0.0665, "num_tokens": 83433.0, "reward": 0.6314187049865723, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.07559289038181305, "reward_meter_mean": 0.7916162014007568, "reward_meter_std": 0.1887982189655304, "reward_std": 0.3165914714336395, "reward_total_composite_mean": 0.6314187049865723, "reward_total_composite_std": 0.3165915012359619, "reward_total_mean": 0.6314187049865723, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.07559289038181305, "rewards/meter/mean": 0.7916162014007568, "rewards/meter/std": 0.1887982189655304, "rewards/total_composite/mean": 0.6314187049865723, "rewards/total_composite/std": 0.3165915012359619, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0329089164733887, "sampling/importance_sampling_ratio/min": 0.1791851669549942, "sampling/sampling_logp_difference/max": 1.7193355560302734, "sampling/sampling_logp_difference/mean": 0.19484373927116394, "step": 41 }, { "clip_ratio/high_max": 0.09933997690677643, "clip_ratio/high_mean": 0.09933997690677643, "clip_ratio/low_mean": 0.12379511073231697, "clip_ratio/low_min": 0.12379511073231697, "clip_ratio/region_mean": 0.2231350876390934, "completions/clipped_ratio": 0.0, "completions/max_length": 100.0, "completions/max_terminated_length": 100.0, "completions/mean_length": 71.5, "completions/mean_terminated_length": 71.5, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 2.382046639919281, "epoch": 0.0016218721037998146, "frac_reward_zero_std": 0.0, "grad_norm": 13.098201751708984, "learning_rate": 9.875757575757576e-06, "loss": 0.1992, "num_tokens": 85365.0, "reward": 0.384579598903656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.384579598903656, "reward_meter_std": 0.38699981570243835, "reward_std": 0.38699978590011597, "reward_total_composite_mean": 0.384579598903656, "reward_total_composite_std": 0.38699981570243835, "reward_total_mean": 0.384579598903656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.384579598903656, "rewards/meter/std": 0.38699981570243835, "rewards/total_composite/mean": 0.384579598903656, "rewards/total_composite/std": 0.38699981570243835, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0288784503936768, "sampling/importance_sampling_ratio/min": 0.11336687952280045, "sampling/sampling_logp_difference/max": 2.177125930786133, "sampling/sampling_logp_difference/mean": 0.216371089220047, "step": 42 }, { "clip_ratio/high_max": 0.07029963098466396, "clip_ratio/high_mean": 0.07029963098466396, "clip_ratio/low_mean": 0.14910552836954594, "clip_ratio/low_min": 0.14910552836954594, "clip_ratio/region_mean": 0.2194051593542099, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 52.875, "completions/mean_terminated_length": 52.875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 1.689612329006195, "epoch": 0.0016604881062712389, "frac_reward_zero_std": 0.0, "grad_norm": 19.66609764099121, "learning_rate": 9.872727272727274e-06, "loss": -0.0628, "num_tokens": 87028.0, "reward": 0.40058913826942444, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.40058913826942444, "reward_meter_std": 0.4409148097038269, "reward_std": 0.4409147799015045, "reward_total_composite_mean": 0.40058913826942444, "reward_total_composite_std": 0.4409148097038269, "reward_total_mean": 0.40058913826942444, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.40058913826942444, "rewards/meter/std": 0.4409148097038269, "rewards/total_composite/mean": 0.40058913826942444, "rewards/total_composite/std": 0.4409148097038269, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0391393899917603, "sampling/importance_sampling_ratio/min": 0.2228635996580124, "sampling/sampling_logp_difference/max": 1.5011954307556152, "sampling/sampling_logp_difference/mean": 0.22723107039928436, "step": 43 }, { "clip_ratio/high_max": 0.041567519307136536, "clip_ratio/high_mean": 0.041567519307136536, "clip_ratio/low_mean": 0.0339520201086998, "clip_ratio/low_min": 0.0339520201086998, "clip_ratio/region_mean": 0.07551953941583633, "completions/clipped_ratio": 0.0, "completions/max_length": 50.0, "completions/max_terminated_length": 50.0, "completions/mean_length": 45.375, "completions/mean_terminated_length": 45.375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.40569959953427315, "epoch": 0.001699104108742663, "frac_reward_zero_std": 0.0, "grad_norm": 20.52503204345703, "learning_rate": 9.869696969696971e-06, "loss": 0.0541, "num_tokens": 88767.0, "reward": 0.6814202070236206, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6814202070236206, "reward_meter_std": 0.4100963771343231, "reward_std": 0.4100963771343231, "reward_total_composite_mean": 0.6814202070236206, "reward_total_composite_std": 0.4100963771343231, "reward_total_mean": 0.6814202070236206, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6814202070236206, "rewards/meter/std": 0.4100963771343231, "rewards/total_composite/mean": 0.6814202070236206, "rewards/total_composite/std": 0.4100963771343231, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9983127117156982, "sampling/importance_sampling_ratio/min": 0.18328185379505157, "sampling/sampling_logp_difference/max": 1.696730136871338, "sampling/sampling_logp_difference/mean": 0.10806272178888321, "step": 44 }, { "clip_ratio/high_max": 0.12581374496221542, "clip_ratio/high_mean": 0.12581374496221542, "clip_ratio/low_mean": 0.07070140354335308, "clip_ratio/low_min": 0.07070140354335308, "clip_ratio/region_mean": 0.1965151485055685, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 48.625, "completions/mean_terminated_length": 48.625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 2.1141539961099625, "epoch": 0.001737720111214087, "frac_reward_zero_std": 0.0, "grad_norm": 19.593364715576172, "learning_rate": 9.866666666666668e-06, "loss": 0.2204, "num_tokens": 90628.0, "reward": 0.6895775198936462, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6895775198936462, "reward_meter_std": 0.43211013078689575, "reward_std": 0.43211016058921814, "reward_total_composite_mean": 0.6895775198936462, "reward_total_composite_std": 0.43211013078689575, "reward_total_mean": 0.6895775198936462, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6895775198936462, "rewards/meter/std": 0.43211013078689575, "rewards/total_composite/mean": 0.6895775198936462, "rewards/total_composite/std": 0.43211013078689575, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0464413166046143, "sampling/importance_sampling_ratio/min": 0.21060624718666077, "sampling/sampling_logp_difference/max": 1.557765007019043, "sampling/sampling_logp_difference/mean": 0.22784224152565002, "step": 45 }, { "clip_ratio/high_max": 0.09895163122564554, "clip_ratio/high_mean": 0.09895163122564554, "clip_ratio/low_mean": 0.11509755253791809, "clip_ratio/low_min": 0.11509755253791809, "clip_ratio/region_mean": 0.21404918376356363, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 36.0, "completions/mean_terminated_length": 36.0, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 1.3433509916067123, "epoch": 0.0017763361136855112, "frac_reward_zero_std": 0.0, "grad_norm": 24.487468719482422, "learning_rate": 9.863636363636364e-06, "loss": 0.0451, "num_tokens": 92020.0, "reward": 0.2181696593761444, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.22091448307037354, "reward_meter_std": 0.17998214066028595, "reward_std": 0.18358123302459717, "reward_total_composite_mean": 0.2181696593761444, "reward_total_composite_std": 0.18358123302459717, "reward_total_mean": 0.2181696593761444, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.22091448307037354, "rewards/meter/std": 0.17998214066028595, "rewards/total_composite/mean": 0.2181696593761444, "rewards/total_composite/std": 0.18358123302459717, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9994174242019653, "sampling/importance_sampling_ratio/min": 0.17236579954624176, "sampling/sampling_logp_difference/max": 1.75813627243042, "sampling/sampling_logp_difference/mean": 0.22377783060073853, "step": 46 }, { "clip_ratio/high_max": 0.0347222238779068, "clip_ratio/high_mean": 0.0347222238779068, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0347222238779068, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 25.5, "completions/mean_terminated_length": 25.5, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 0.1728968620300293, "epoch": 0.0018149521161569355, "frac_reward_zero_std": 0.0, "grad_norm": 19.872201919555664, "learning_rate": 9.860606060606061e-06, "loss": -0.0284, "num_tokens": 93400.0, "reward": 0.8368544578552246, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8368544578552246, "reward_meter_std": 0.035664137452840805, "reward_std": 0.0356641449034214, "reward_total_composite_mean": 0.8368544578552246, "reward_total_composite_std": 0.035664137452840805, "reward_total_mean": 0.8368544578552246, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8368544578552246, "rewards/meter/std": 0.035664137452840805, "rewards/total_composite/mean": 0.8368544578552246, "rewards/total_composite/std": 0.035664137452840805, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9922424554824829, "sampling/importance_sampling_ratio/min": 0.0397215262055397, "sampling/sampling_logp_difference/max": 3.2258620262145996, "sampling/sampling_logp_difference/mean": 0.07806604355573654, "step": 47 }, { "clip_ratio/high_max": 0.13037441484630108, "clip_ratio/high_mean": 0.13037441484630108, "clip_ratio/low_mean": 0.09916602075099945, "clip_ratio/low_min": 0.09916602075099945, "clip_ratio/region_mean": 0.22954043559730053, "completions/clipped_ratio": 0.0, "completions/max_length": 136.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 95.125, "completions/mean_terminated_length": 95.125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 2.133227914571762, "epoch": 0.0018535681186283596, "frac_reward_zero_std": 0.0, "grad_norm": 13.035599708557129, "learning_rate": 9.857575757575758e-06, "loss": 0.0975, "num_tokens": 95529.0, "reward": 0.3589945435523987, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3589945435523987, "reward_meter_std": 0.3197920322418213, "reward_std": 0.3197920024394989, "reward_total_composite_mean": 0.3589945435523987, "reward_total_composite_std": 0.3197920322418213, "reward_total_mean": 0.3589945435523987, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3589945435523987, "rewards/meter/std": 0.3197920322418213, "rewards/total_composite/mean": 0.3589945435523987, "rewards/total_composite/std": 0.3197920322418213, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0021162033081055, "sampling/importance_sampling_ratio/min": 0.12261507660150528, "sampling/sampling_logp_difference/max": 2.098705291748047, "sampling/sampling_logp_difference/mean": 0.23110564053058624, "step": 48 }, { "clip_ratio/high_max": 0.13691459875553846, "clip_ratio/high_mean": 0.13691459875553846, "clip_ratio/low_mean": 0.0817372314631939, "clip_ratio/low_min": 0.0817372314631939, "clip_ratio/region_mean": 0.21865183021873236, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 48.5, "completions/mean_terminated_length": 48.5, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 2.332765683531761, "epoch": 0.0018921841210997837, "frac_reward_zero_std": 0.0, "grad_norm": 15.745153427124023, "learning_rate": 9.854545454545456e-06, "loss": 0.0542, "num_tokens": 97325.0, "reward": 0.5693104267120361, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5693104267120361, "reward_meter_std": 0.4606866240501404, "reward_std": 0.460686594247818, "reward_total_composite_mean": 0.5693104267120361, "reward_total_composite_std": 0.4606866240501404, "reward_total_mean": 0.5693104267120361, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5693104267120361, "rewards/meter/std": 0.4606866240501404, "rewards/total_composite/mean": 0.5693104267120361, "rewards/total_composite/std": 0.4606866240501404, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.031612515449524, "sampling/importance_sampling_ratio/min": 0.15585720539093018, "sampling/sampling_logp_difference/max": 1.8588149547576904, "sampling/sampling_logp_difference/mean": 0.2219070941209793, "step": 49 }, { "clip_ratio/high_max": 0.13284815661609173, "clip_ratio/high_mean": 0.13284815661609173, "clip_ratio/low_mean": 0.06395815405994654, "clip_ratio/low_min": 0.06395815405994654, "clip_ratio/region_mean": 0.19680631067603827, "completions/clipped_ratio": 0.0, "completions/max_length": 214.0, "completions/max_terminated_length": 214.0, "completions/mean_length": 176.625, "completions/mean_terminated_length": 176.625, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 2.019968628883362, "epoch": 0.0019308001235712078, "frac_reward_zero_std": 0.0, "grad_norm": 7.936125755310059, "learning_rate": 9.851515151515151e-06, "loss": -0.0192, "num_tokens": 100154.0, "reward": 0.5725899934768677, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.7158485651016235, "reward_meter_std": 0.3752211630344391, "reward_std": 0.41487547755241394, "reward_total_composite_mean": 0.5725899934768677, "reward_total_composite_std": 0.41487547755241394, "reward_total_mean": 0.5725899934768677, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.7158485651016235, "rewards/meter/std": 0.3752211630344391, "rewards/total_composite/mean": 0.5725899934768677, "rewards/total_composite/std": 0.41487547755241394, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.021142601966858, "sampling/importance_sampling_ratio/min": 0.10096535831689835, "sampling/sampling_logp_difference/max": 2.292977809906006, "sampling/sampling_logp_difference/mean": 0.19521863758563995, "step": 50 }, { "epoch": 0.0019308001235712078, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/max_length": 426.38461538461536, "eval_completions/max_terminated_length": 346.46153846153845, "eval_completions/mean_length": 184.7403846153846, "eval_completions/mean_terminated_length": 164.79945608285757, "eval_completions/min_length": 49.30769230769231, "eval_completions/min_terminated_length": 49.30769230769231, "eval_entropy": 2.3565903168458204, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 100154.0, "eval_reward": 0.3838638663291931, "eval_reward_arabic_clean_mean": 0.9230769230769231, "eval_reward_arabic_clean_std": 0.1987869510283837, "eval_reward_count_adherence_mean": 0.9684277910452622, "eval_reward_count_adherence_std": 0.059289679647638246, "eval_reward_meter_mean": 0.4141139342234685, "eval_reward_meter_std": 0.3737386694321266, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.3838638663291931, "eval_reward_total_composite_std": 0.3831195647899921, "eval_reward_total_mean": 0.3838638663291931, "eval_rewards/arabic_clean/mean": 0.9230769230769231, "eval_rewards/arabic_clean/std": 0.1987869510283837, "eval_rewards/count_adherence/mean": 0.9684277910452622, "eval_rewards/count_adherence/std": 0.059289679647638246, "eval_rewards/meter/mean": 0.4141139342234685, "eval_rewards/meter/std": 0.3737386694321266, "eval_rewards/total_composite/mean": 0.3838638663291931, "eval_rewards/total_composite/std": 0.3831195647899921, "eval_runtime": 79.1676, "eval_samples_per_second": 1.314, "eval_sampling/importance_sampling_ratio/max": 1.6281351492955134, "eval_sampling/importance_sampling_ratio/mean": 1.0421396860709558, "eval_sampling/importance_sampling_ratio/min": 0.31346100110274094, "eval_sampling/sampling_logp_difference/max": 1.170855081998385, "eval_sampling/sampling_logp_difference/mean": 0.14078256086661264, "eval_steps_per_second": 0.164, "step": 50 }, { "clip_ratio/high_max": 0.06337687559425831, "clip_ratio/high_mean": 0.06337687559425831, "clip_ratio/low_mean": 0.16671104542911053, "clip_ratio/low_min": 0.16671104542911053, "clip_ratio/region_mean": 0.23008792102336884, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 41.875, "completions/mean_terminated_length": 41.875, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 1.8089727088809013, "epoch": 0.001969416126042632, "frac_reward_zero_std": 0.0, "grad_norm": 18.508319854736328, "learning_rate": 9.84848484848485e-06, "loss": 0.0165, "num_tokens": 101753.0, "reward": 0.23897752165794373, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.23897752165794373, "reward_meter_std": 0.3632267415523529, "reward_std": 0.3632267415523529, "reward_total_composite_mean": 0.23897752165794373, "reward_total_composite_std": 0.3632267415523529, "reward_total_mean": 0.23897752165794373, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.23897752165794373, "rewards/meter/std": 0.3632267415523529, "rewards/total_composite/mean": 0.23897752165794373, "rewards/total_composite/std": 0.3632267415523529, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0173118114471436, "sampling/importance_sampling_ratio/min": 0.08802343904972076, "sampling/sampling_logp_difference/max": 2.430152177810669, "sampling/sampling_logp_difference/mean": 0.2588263154029846, "step": 51 }, { "clip_ratio/high_max": 0.14840957894921303, "clip_ratio/high_mean": 0.14840957894921303, "clip_ratio/low_mean": 0.048903508111834526, "clip_ratio/low_min": 0.048903508111834526, "clip_ratio/region_mean": 0.19731308706104755, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 67.375, "completions/mean_terminated_length": 67.375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 2.2769337445497513, "epoch": 0.002008032128514056, "frac_reward_zero_std": 0.0, "grad_norm": 11.88036060333252, "learning_rate": 9.845454545454546e-06, "loss": -0.048, "num_tokens": 103684.0, "reward": 0.9933522939682007, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9933522939682007, "reward_meter_std": 0.004396223928779364, "reward_std": 0.004396222531795502, "reward_total_composite_mean": 0.9933522939682007, "reward_total_composite_std": 0.004396223928779364, "reward_total_mean": 0.9933522939682007, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9933522939682007, "rewards/meter/std": 0.004396223928779364, "rewards/total_composite/mean": 0.9933522939682007, "rewards/total_composite/std": 0.004396223928779364, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0523029565811157, "sampling/importance_sampling_ratio/min": 0.22250288724899292, "sampling/sampling_logp_difference/max": 1.5028152465820312, "sampling/sampling_logp_difference/mean": 0.20311212539672852, "step": 52 }, { "clip_ratio/high_max": 0.1322573497891426, "clip_ratio/high_mean": 0.1322573497891426, "clip_ratio/low_mean": 0.06783267110586166, "clip_ratio/low_min": 0.06783267110586166, "clip_ratio/region_mean": 0.20009002089500427, "completions/clipped_ratio": 0.0, "completions/max_length": 52.0, "completions/max_terminated_length": 52.0, "completions/mean_length": 37.125, "completions/mean_terminated_length": 37.125, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 1.6774490028619766, "epoch": 0.0020466481309854806, "frac_reward_zero_std": 0.0, "grad_norm": 19.141048431396484, "learning_rate": 9.842424242424243e-06, "loss": 0.0962, "num_tokens": 105125.0, "reward": 0.5976195335388184, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5984563231468201, "reward_meter_std": 0.42879587411880493, "reward_std": 0.43012017011642456, "reward_total_composite_mean": 0.5976195335388184, "reward_total_composite_std": 0.43012017011642456, "reward_total_mean": 0.5976195335388184, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5984563231468201, "rewards/meter/std": 0.42879587411880493, "rewards/total_composite/mean": 0.5976195335388184, "rewards/total_composite/std": 0.43012017011642456, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0220152139663696, "sampling/importance_sampling_ratio/min": 0.255797803401947, "sampling/sampling_logp_difference/max": 1.363368034362793, "sampling/sampling_logp_difference/mean": 0.2187996506690979, "step": 53 }, { "clip_ratio/high_max": 0.12507456727325916, "clip_ratio/high_mean": 0.12507456727325916, "clip_ratio/low_mean": 0.08917682990431786, "clip_ratio/low_min": 0.08917682990431786, "clip_ratio/region_mean": 0.21425139717757702, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 52.125, "completions/mean_terminated_length": 52.125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 2.4166207313537598, "epoch": 0.0020852641334569047, "frac_reward_zero_std": 0.0, "grad_norm": 15.735599517822266, "learning_rate": 9.83939393939394e-06, "loss": -0.066, "num_tokens": 106862.0, "reward": 0.6028301119804382, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6028301119804382, "reward_meter_std": 0.4117062985897064, "reward_std": 0.41170626878738403, "reward_total_composite_mean": 0.6028301119804382, "reward_total_composite_std": 0.4117062985897064, "reward_total_mean": 0.6028301119804382, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6028301119804382, "rewards/meter/std": 0.4117062985897064, "rewards/total_composite/mean": 0.6028301119804382, "rewards/total_composite/std": 0.4117062985897064, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0488487482070923, "sampling/importance_sampling_ratio/min": 0.19384431838989258, "sampling/sampling_logp_difference/max": 1.640699863433838, "sampling/sampling_logp_difference/mean": 0.21817506849765778, "step": 54 }, { "clip_ratio/high_max": 0.07919401116669178, "clip_ratio/high_mean": 0.07919401116669178, "clip_ratio/low_mean": 0.1296012494713068, "clip_ratio/low_min": 0.1296012494713068, "clip_ratio/region_mean": 0.20879526063799858, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 57.875, "completions/mean_terminated_length": 57.875, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 2.436240643262863, "epoch": 0.002123880135928329, "frac_reward_zero_std": 0.0, "grad_norm": 16.236080169677734, "learning_rate": 9.836363636363637e-06, "loss": 0.0736, "num_tokens": 108605.0, "reward": 0.2679245173931122, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2679245173931122, "reward_meter_std": 0.29882389307022095, "reward_std": 0.29882386326789856, "reward_total_composite_mean": 0.2679245173931122, "reward_total_composite_std": 0.29882389307022095, "reward_total_mean": 0.2679245173931122, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2679245173931122, "rewards/meter/std": 0.29882389307022095, "rewards/total_composite/mean": 0.2679245173931122, "rewards/total_composite/std": 0.29882389307022095, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0213844776153564, "sampling/importance_sampling_ratio/min": 0.13746440410614014, "sampling/sampling_logp_difference/max": 1.9843902587890625, "sampling/sampling_logp_difference/mean": 0.2685631215572357, "step": 55 }, { "clip_ratio/high_max": 0.09804818406701088, "clip_ratio/high_mean": 0.09804818406701088, "clip_ratio/low_mean": 0.054613095708191395, "clip_ratio/low_min": 0.054613095708191395, "clip_ratio/region_mean": 0.15266127977520227, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 42.75, "completions/mean_terminated_length": 42.75, "completions/min_length": 24.0, "completions/min_terminated_length": 24.0, "entropy": 1.5369336754083633, "epoch": 0.002162496138399753, "frac_reward_zero_std": 0.0, "grad_norm": 17.529335021972656, "learning_rate": 9.833333333333333e-06, "loss": -0.1085, "num_tokens": 110163.0, "reward": 0.6618984937667847, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7218412160873413, "reward_meter_std": 0.4222549796104431, "reward_std": 0.4177789092063904, "reward_total_composite_mean": 0.6618984937667847, "reward_total_composite_std": 0.41777893900871277, "reward_total_mean": 0.6618984937667847, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7218412160873413, "rewards/meter/std": 0.4222549796104431, "rewards/total_composite/mean": 0.6618984937667847, "rewards/total_composite/std": 0.41777893900871277, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0329205989837646, "sampling/importance_sampling_ratio/min": 0.1972530037164688, "sampling/sampling_logp_difference/max": 1.8073515892028809, "sampling/sampling_logp_difference/mean": 0.20614919066429138, "step": 56 }, { "clip_ratio/high_max": 0.1381522510200739, "clip_ratio/high_mean": 0.1381522510200739, "clip_ratio/low_mean": 0.07145614549517632, "clip_ratio/low_min": 0.07145614549517632, "clip_ratio/region_mean": 0.2096083965152502, "completions/clipped_ratio": 0.0, "completions/max_length": 266.0, "completions/max_terminated_length": 266.0, "completions/mean_length": 211.875, "completions/mean_terminated_length": 211.875, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 3.097422182559967, "epoch": 0.002201112140871177, "frac_reward_zero_std": 0.0, "grad_norm": 7.2061991691589355, "learning_rate": 9.830303030303032e-06, "loss": 0.0427, "num_tokens": 113546.0, "reward": 0.7174510955810547, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.8386244773864746, "reward_meter_std": 0.1593872308731079, "reward_std": 0.33248570561408997, "reward_total_composite_mean": 0.7174510955810547, "reward_total_composite_std": 0.33248570561408997, "reward_total_mean": 0.7174510955810547, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.8386244773864746, "rewards/meter/std": 0.1593872308731079, "rewards/total_composite/mean": 0.7174510955810547, "rewards/total_composite/std": 0.33248570561408997, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0385682582855225, "sampling/importance_sampling_ratio/min": 0.17200742661952972, "sampling/sampling_logp_difference/max": 1.7602176666259766, "sampling/sampling_logp_difference/mean": 0.2227792590856552, "step": 57 }, { "clip_ratio/high_max": 0.20505854487419128, "clip_ratio/high_mean": 0.20505854487419128, "clip_ratio/low_mean": 0.01875000074505806, "clip_ratio/low_min": 0.01875000074505806, "clip_ratio/region_mean": 0.22380854561924934, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 2.203368306159973, "epoch": 0.002239728143342601, "frac_reward_zero_std": 0.0, "grad_norm": 10.99088191986084, "learning_rate": 9.827272727272729e-06, "loss": 0.0085, "num_tokens": 115430.0, "reward": 0.8916192054748535, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8916192054748535, "reward_meter_std": 0.292441725730896, "reward_std": 0.292441725730896, "reward_total_composite_mean": 0.8916192054748535, "reward_total_composite_std": 0.292441725730896, "reward_total_mean": 0.8916192054748535, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8916192054748535, "rewards/meter/std": 0.292441725730896, "rewards/total_composite/mean": 0.8916192054748535, "rewards/total_composite/std": 0.292441725730896, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0151267051696777, "sampling/importance_sampling_ratio/min": 0.0522882305085659, "sampling/sampling_logp_difference/max": 2.950984001159668, "sampling/sampling_logp_difference/mean": 0.2055407166481018, "step": 58 }, { "clip_ratio/high_max": 0.07533212564885616, "clip_ratio/high_mean": 0.07533212564885616, "clip_ratio/low_mean": 0.13746320828795433, "clip_ratio/low_min": 0.13746320828795433, "clip_ratio/region_mean": 0.2127953339368105, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 100.5, "completions/mean_terminated_length": 100.5, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 2.076779752969742, "epoch": 0.002278344145814025, "frac_reward_zero_std": 0.0, "grad_norm": 13.52698040008545, "learning_rate": 9.824242424242425e-06, "loss": 0.1166, "num_tokens": 117834.0, "reward": 0.3059839606285095, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.30988091230392456, "reward_meter_std": 0.22690658271312714, "reward_std": 0.2295694798231125, "reward_total_composite_mean": 0.3059839606285095, "reward_total_composite_std": 0.22956949472427368, "reward_total_mean": 0.3059839606285095, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.30988091230392456, "rewards/meter/std": 0.22690658271312714, "rewards/total_composite/mean": 0.3059839606285095, "rewards/total_composite/std": 0.22956949472427368, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9983330368995667, "sampling/importance_sampling_ratio/min": 0.08552319556474686, "sampling/sampling_logp_difference/max": 2.458967685699463, "sampling/sampling_logp_difference/mean": 0.2522209882736206, "step": 59 }, { "clip_ratio/high_max": 0.11208062618970871, "clip_ratio/high_mean": 0.11208062618970871, "clip_ratio/low_mean": 0.12309817224740982, "clip_ratio/low_min": 0.12309817224740982, "clip_ratio/region_mean": 0.23517879843711853, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 53.125, "completions/mean_terminated_length": 53.125, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 2.3633119463920593, "epoch": 0.0023169601482854493, "frac_reward_zero_std": 0.0, "grad_norm": 13.342586517333984, "learning_rate": 9.821212121212122e-06, "loss": 0.0991, "num_tokens": 119499.0, "reward": 0.3410363793373108, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3410363793373108, "reward_meter_std": 0.28698158264160156, "reward_std": 0.28698158264160156, "reward_total_composite_mean": 0.3410363793373108, "reward_total_composite_std": 0.28698158264160156, "reward_total_mean": 0.3410363793373108, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3410363793373108, "rewards/meter/std": 0.28698158264160156, "rewards/total_composite/mean": 0.3410363793373108, "rewards/total_composite/std": 0.28698158264160156, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0315275192260742, "sampling/importance_sampling_ratio/min": 0.2565752863883972, "sampling/sampling_logp_difference/max": 1.3603332042694092, "sampling/sampling_logp_difference/mean": 0.21170011162757874, "step": 60 }, { "clip_ratio/high_max": 0.17636545840650797, "clip_ratio/high_mean": 0.17636545840650797, "clip_ratio/low_mean": 0.036764707416296005, "clip_ratio/low_min": 0.036764707416296005, "clip_ratio/region_mean": 0.21313016582280397, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 29.75, "completions/mean_terminated_length": 29.75, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 2.489090621471405, "epoch": 0.002355576150756874, "frac_reward_zero_std": 0.0, "grad_norm": 25.383638381958008, "learning_rate": 9.81818181818182e-06, "loss": 0.062, "num_tokens": 121057.0, "reward": 0.9133638143539429, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9133638143539429, "reward_meter_std": 0.2053176313638687, "reward_std": 0.2053176462650299, "reward_total_composite_mean": 0.9133638143539429, "reward_total_composite_std": 0.2053176313638687, "reward_total_mean": 0.9133638143539429, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9133638143539429, "rewards/meter/std": 0.2053176313638687, "rewards/total_composite/mean": 0.9133638143539429, "rewards/total_composite/std": 0.2053176313638687, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0374253988265991, "sampling/importance_sampling_ratio/min": 0.25975415110588074, "sampling/sampling_logp_difference/max": 1.3480195999145508, "sampling/sampling_logp_difference/mean": 0.19405041635036469, "step": 61 }, { "clip_ratio/high_max": 0.09878450445830822, "clip_ratio/high_mean": 0.09878450445830822, "clip_ratio/low_mean": 0.100760068744421, "clip_ratio/low_min": 0.100760068744421, "clip_ratio/region_mean": 0.19954457320272923, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 99.375, "completions/mean_terminated_length": 99.375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 2.4795138239860535, "epoch": 0.002394192153228298, "frac_reward_zero_std": 0.0, "grad_norm": 12.204435348510742, "learning_rate": 9.815151515151516e-06, "loss": 0.1325, "num_tokens": 123204.0, "reward": 0.4881056547164917, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4881056547164917, "reward_meter_std": 0.2547048330307007, "reward_std": 0.2547048032283783, "reward_total_composite_mean": 0.4881056547164917, "reward_total_composite_std": 0.2547048330307007, "reward_total_mean": 0.4881056547164917, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4881056547164917, "rewards/meter/std": 0.2547048330307007, "rewards/total_composite/mean": 0.4881056547164917, "rewards/total_composite/std": 0.2547048330307007, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.041177749633789, "sampling/importance_sampling_ratio/min": 0.15324591100215912, "sampling/sampling_logp_difference/max": 1.875711441040039, "sampling/sampling_logp_difference/mean": 0.21773025393486023, "step": 62 }, { "clip_ratio/high_max": 0.1503895577043295, "clip_ratio/high_mean": 0.1503895577043295, "clip_ratio/low_mean": 0.02020994247868657, "clip_ratio/low_min": 0.02020994247868657, "clip_ratio/region_mean": 0.17059950018301606, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 60.875, "completions/mean_terminated_length": 60.875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 2.3204995840787888, "epoch": 0.002432808155699722, "frac_reward_zero_std": 0.0, "grad_norm": 18.67860221862793, "learning_rate": 9.812121212121212e-06, "loss": 0.0373, "num_tokens": 124939.0, "reward": 0.9198144674301147, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9198144674301147, "reward_meter_std": 0.13751091063022614, "reward_std": 0.13751091063022614, "reward_total_composite_mean": 0.9198144674301147, "reward_total_composite_std": 0.13751091063022614, "reward_total_mean": 0.9198144674301147, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9198144674301147, "rewards/meter/std": 0.13751091063022614, "rewards/total_composite/mean": 0.9198144674301147, "rewards/total_composite/std": 0.13751091063022614, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0526179075241089, "sampling/importance_sampling_ratio/min": 0.23382781445980072, "sampling/sampling_logp_difference/max": 1.4531702995300293, "sampling/sampling_logp_difference/mean": 0.19855335354804993, "step": 63 }, { "clip_ratio/high_max": 0.1345098316669464, "clip_ratio/high_mean": 0.1345098316669464, "clip_ratio/low_mean": 0.06851893290877342, "clip_ratio/low_min": 0.06851893290877342, "clip_ratio/region_mean": 0.20302876457571983, "completions/clipped_ratio": 0.0, "completions/max_length": 410.0, "completions/max_terminated_length": 410.0, "completions/mean_length": 308.875, "completions/mean_terminated_length": 308.875, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 3.1392965018749237, "epoch": 0.002471424158171146, "frac_reward_zero_std": 0.0, "grad_norm": 5.2523932456970215, "learning_rate": 9.809090909090911e-06, "loss": -0.052, "num_tokens": 129130.0, "reward": 0.5670943260192871, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.05175492912530899, "reward_meter_mean": 0.9247400164604187, "reward_meter_std": 0.13669492304325104, "reward_std": 0.47056710720062256, "reward_total_composite_mean": 0.5670943260192871, "reward_total_composite_std": 0.47056710720062256, "reward_total_mean": 0.5670943260192871, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.05175492912530899, "rewards/meter/mean": 0.9247400164604187, "rewards/meter/std": 0.13669492304325104, "rewards/total_composite/mean": 0.5670943260192871, "rewards/total_composite/std": 0.47056710720062256, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0358939170837402, "sampling/importance_sampling_ratio/min": 0.15853774547576904, "sampling/sampling_logp_difference/max": 1.8417625427246094, "sampling/sampling_logp_difference/mean": 0.21920810639858246, "step": 64 }, { "clip_ratio/high_max": 0.08198772463947535, "clip_ratio/high_mean": 0.08198772463947535, "clip_ratio/low_mean": 0.11948978900909424, "clip_ratio/low_min": 0.11948978900909424, "clip_ratio/region_mean": 0.20147751364856958, "completions/clipped_ratio": 0.0, "completions/max_length": 48.0, "completions/max_terminated_length": 48.0, "completions/mean_length": 35.875, "completions/mean_terminated_length": 35.875, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 1.4370453506708145, "epoch": 0.0025100401606425703, "frac_reward_zero_std": 0.0, "grad_norm": 26.357711791992188, "learning_rate": 9.806060606060607e-06, "loss": 0.2664, "num_tokens": 130721.0, "reward": 0.5192750096321106, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5192750096321106, "reward_meter_std": 0.47312480211257935, "reward_std": 0.47312480211257935, "reward_total_composite_mean": 0.5192750096321106, "reward_total_composite_std": 0.47312480211257935, "reward_total_mean": 0.5192750096321106, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5192750096321106, "rewards/meter/std": 0.47312480211257935, "rewards/total_composite/mean": 0.5192750096321106, "rewards/total_composite/std": 0.47312480211257935, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0055742263793945, "sampling/importance_sampling_ratio/min": 0.13099518418312073, "sampling/sampling_logp_difference/max": 2.032594680786133, "sampling/sampling_logp_difference/mean": 0.2368556261062622, "step": 65 }, { "clip_ratio/high_max": 0.13170645385980606, "clip_ratio/high_mean": 0.13170645385980606, "clip_ratio/low_mean": 0.046164773404598236, "clip_ratio/low_min": 0.046164773404598236, "clip_ratio/region_mean": 0.1778712272644043, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 1.377866268157959, "epoch": 0.0025486561631139944, "frac_reward_zero_std": 0.0, "grad_norm": 27.630306243896484, "learning_rate": 9.803030303030304e-06, "loss": 0.137, "num_tokens": 132488.0, "reward": 0.7682692408561707, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7682692408561707, "reward_meter_std": 0.3781956434249878, "reward_std": 0.3781956434249878, "reward_total_composite_mean": 0.7682692408561707, "reward_total_composite_std": 0.3781956434249878, "reward_total_mean": 0.7682692408561707, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7682692408561707, "rewards/meter/std": 0.3781956434249878, "rewards/total_composite/mean": 0.7682692408561707, "rewards/total_composite/std": 0.3781956434249878, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000638723373413, "sampling/importance_sampling_ratio/min": 0.1309218853712082, "sampling/sampling_logp_difference/max": 2.0331544876098633, "sampling/sampling_logp_difference/mean": 0.21417810022830963, "step": 66 }, { "clip_ratio/high_max": 0.13258693367242813, "clip_ratio/high_mean": 0.13258693367242813, "clip_ratio/low_mean": 0.08497239649295807, "clip_ratio/low_min": 0.08497239649295807, "clip_ratio/region_mean": 0.2175593301653862, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 48.875, "completions/mean_terminated_length": 48.875, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 2.8289676904678345, "epoch": 0.0025872721655854185, "frac_reward_zero_std": 0.0, "grad_norm": 14.999178886413574, "learning_rate": 9.800000000000001e-06, "loss": 0.0209, "num_tokens": 134295.0, "reward": 0.6665328741073608, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.8183262348175049, "reward_meter_std": 0.32211264967918396, "reward_std": 0.43555328249931335, "reward_total_composite_mean": 0.6665328741073608, "reward_total_composite_std": 0.43555328249931335, "reward_total_mean": 0.6665328741073608, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.8183262348175049, "rewards/meter/std": 0.32211264967918396, "rewards/total_composite/mean": 0.6665328741073608, "rewards/total_composite/std": 0.43555328249931335, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0471768379211426, "sampling/importance_sampling_ratio/min": 0.23626382648944855, "sampling/sampling_logp_difference/max": 1.4428062438964844, "sampling/sampling_logp_difference/mean": 0.2460046112537384, "step": 67 }, { "clip_ratio/high_max": 0.10875347442924976, "clip_ratio/high_mean": 0.10875347442924976, "clip_ratio/low_mean": 0.08730671741068363, "clip_ratio/low_min": 0.08730671741068363, "clip_ratio/region_mean": 0.1960601918399334, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 52.75, "completions/mean_terminated_length": 52.75, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 1.6037258356809616, "epoch": 0.0026258881680568426, "frac_reward_zero_std": 0.0, "grad_norm": 15.564022064208984, "learning_rate": 9.796969696969698e-06, "loss": 0.1484, "num_tokens": 135997.0, "reward": 0.35390323400497437, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4771120548248291, "reward_meter_std": 0.3203728497028351, "reward_std": 0.28436198830604553, "reward_total_composite_mean": 0.35390323400497437, "reward_total_composite_std": 0.28436195850372314, "reward_total_mean": 0.35390323400497437, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4771120548248291, "rewards/meter/std": 0.3203728497028351, "rewards/total_composite/mean": 0.35390323400497437, "rewards/total_composite/std": 0.28436195850372314, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0302598476409912, "sampling/importance_sampling_ratio/min": 0.23306208848953247, "sampling/sampling_logp_difference/max": 1.4564504623413086, "sampling/sampling_logp_difference/mean": 0.22163131833076477, "step": 68 }, { "clip_ratio/high_max": 0.15749920904636383, "clip_ratio/high_mean": 0.15749920904636383, "clip_ratio/low_mean": 0.06984523870050907, "clip_ratio/low_min": 0.06984523870050907, "clip_ratio/region_mean": 0.2273444477468729, "completions/clipped_ratio": 0.0, "completions/max_length": 147.0, "completions/max_terminated_length": 147.0, "completions/mean_length": 112.5, "completions/mean_terminated_length": 112.5, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 3.092645615339279, "epoch": 0.002664504170528267, "frac_reward_zero_std": 0.0, "grad_norm": 8.40172290802002, "learning_rate": 9.793939393939394e-06, "loss": 0.1465, "num_tokens": 138233.0, "reward": 0.5694236159324646, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6559433937072754, "reward_meter_std": 0.3531401455402374, "reward_std": 0.4212261736392975, "reward_total_composite_mean": 0.5694236159324646, "reward_total_composite_std": 0.4212262034416199, "reward_total_mean": 0.5694236159324646, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6559433937072754, "rewards/meter/std": 0.3531401455402374, "rewards/total_composite/mean": 0.5694236159324646, "rewards/total_composite/std": 0.4212262034416199, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0383771657943726, "sampling/importance_sampling_ratio/min": 0.18941885232925415, "sampling/sampling_logp_difference/max": 1.6637945175170898, "sampling/sampling_logp_difference/mean": 0.21621394157409668, "step": 69 }, { "clip_ratio/high_max": 0.07498700357973576, "clip_ratio/high_mean": 0.07498700357973576, "clip_ratio/low_mean": 0.10936851240694523, "clip_ratio/low_min": 0.10936851240694523, "clip_ratio/region_mean": 0.18435551598668098, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 78.625, "completions/mean_terminated_length": 78.625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 2.6834762394428253, "epoch": 0.0027031201729996912, "frac_reward_zero_std": 0.0, "grad_norm": 12.81651782989502, "learning_rate": 9.790909090909093e-06, "loss": 0.0207, "num_tokens": 140310.0, "reward": 0.4240095019340515, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.43477553129196167, "reward_meter_std": 0.332572340965271, "reward_std": 0.337272584438324, "reward_total_composite_mean": 0.4240095019340515, "reward_total_composite_std": 0.337272584438324, "reward_total_mean": 0.4240095019340515, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.43477553129196167, "rewards/meter/std": 0.332572340965271, "rewards/total_composite/mean": 0.4240095019340515, "rewards/total_composite/std": 0.337272584438324, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.031362533569336, "sampling/importance_sampling_ratio/min": 0.2754969000816345, "sampling/sampling_logp_difference/max": 1.2891788482666016, "sampling/sampling_logp_difference/mean": 0.2150382101535797, "step": 70 }, { "clip_ratio/high_max": 0.12004932574927807, "clip_ratio/high_mean": 0.12004932574927807, "clip_ratio/low_mean": 0.07749529182910919, "clip_ratio/low_min": 0.07749529182910919, "clip_ratio/region_mean": 0.19754461757838726, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 77.5, "completions/mean_terminated_length": 77.5, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 2.1531532555818558, "epoch": 0.0027417361754711153, "frac_reward_zero_std": 0.0, "grad_norm": 13.432793617248535, "learning_rate": 9.787878787878788e-06, "loss": 0.0109, "num_tokens": 142298.0, "reward": 0.7772670984268188, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7772670984268188, "reward_meter_std": 0.3389206826686859, "reward_std": 0.3389207124710083, "reward_total_composite_mean": 0.7772670984268188, "reward_total_composite_std": 0.3389206826686859, "reward_total_mean": 0.7772670984268188, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7772670984268188, "rewards/meter/std": 0.3389206826686859, "rewards/total_composite/mean": 0.7772670984268188, "rewards/total_composite/std": 0.3389206826686859, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0325590372085571, "sampling/importance_sampling_ratio/min": 0.2021116316318512, "sampling/sampling_logp_difference/max": 1.5989351272583008, "sampling/sampling_logp_difference/mean": 0.1868724375963211, "step": 71 }, { "clip_ratio/high_max": 0.11257654242217541, "clip_ratio/high_mean": 0.11257654242217541, "clip_ratio/low_mean": 0.0991472564637661, "clip_ratio/low_min": 0.0991472564637661, "clip_ratio/region_mean": 0.2117237988859415, "completions/clipped_ratio": 0.0, "completions/max_length": 175.0, "completions/max_terminated_length": 175.0, "completions/mean_length": 153.375, "completions/mean_terminated_length": 153.375, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "entropy": 2.355428859591484, "epoch": 0.0027803521779425394, "frac_reward_zero_std": 0.0, "grad_norm": 9.057010650634766, "learning_rate": 9.784848484848486e-06, "loss": 0.1169, "num_tokens": 144949.0, "reward": 0.5220509767532349, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.1414213478565216, "reward_meter_mean": 0.5224494934082031, "reward_meter_std": 0.4397116005420685, "reward_std": 0.4402455985546112, "reward_total_composite_mean": 0.5220509767532349, "reward_total_composite_std": 0.4402455985546112, "reward_total_mean": 0.5220509767532349, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.1414213478565216, "rewards/meter/mean": 0.5224494934082031, "rewards/meter/std": 0.4397116005420685, "rewards/total_composite/mean": 0.5220509767532349, "rewards/total_composite/std": 0.4402455985546112, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0310229063034058, "sampling/importance_sampling_ratio/min": 0.16696415841579437, "sampling/sampling_logp_difference/max": 1.7899761199951172, "sampling/sampling_logp_difference/mean": 0.2253074198961258, "step": 72 }, { "clip_ratio/high_max": 0.11561089940369129, "clip_ratio/high_mean": 0.11561089940369129, "clip_ratio/low_mean": 0.10295584239065647, "clip_ratio/low_min": 0.10295584239065647, "clip_ratio/region_mean": 0.21856674179434776, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 56.875, "completions/mean_terminated_length": 56.875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 2.409619852900505, "epoch": 0.0028189681804139635, "frac_reward_zero_std": 0.0, "grad_norm": 14.417825698852539, "learning_rate": 9.781818181818183e-06, "loss": 0.025, "num_tokens": 146868.0, "reward": 0.40673646330833435, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.44528478384017944, "reward_meter_std": 0.402529776096344, "reward_std": 0.43125417828559875, "reward_total_composite_mean": 0.40673646330833435, "reward_total_composite_std": 0.43125420808792114, "reward_total_mean": 0.40673646330833435, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.44528478384017944, "rewards/meter/std": 0.402529776096344, "rewards/total_composite/mean": 0.40673646330833435, "rewards/total_composite/std": 0.43125420808792114, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008949637413025, "sampling/importance_sampling_ratio/min": 0.16759149730205536, "sampling/sampling_logp_difference/max": 1.7862257957458496, "sampling/sampling_logp_difference/mean": 0.2235013246536255, "step": 73 }, { "clip_ratio/high_max": 0.02222863771021366, "clip_ratio/high_mean": 0.02222863771021366, "clip_ratio/low_mean": 0.05759216099977493, "clip_ratio/low_min": 0.05759216099977493, "clip_ratio/region_mean": 0.0798207987099886, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 465.0, "completions/mean_length": 479.375, "completions/mean_terminated_length": 446.75, "completions/min_length": 426.0, "completions/min_terminated_length": 426.0, "entropy": 1.72447469830513, "epoch": 0.0028575841828853876, "frac_reward_zero_std": 0.0, "grad_norm": 1.7615196704864502, "learning_rate": 9.77878787878788e-06, "loss": -0.011, "num_tokens": 150527.0, "reward": 0.5821076035499573, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8833333253860474, "reward_count_adherence_std": 0.077664315700531, "reward_meter_mean": 0.6778547763824463, "reward_meter_std": 0.2513851225376129, "reward_std": 0.29507479071617126, "reward_total_composite_mean": 0.5821076035499573, "reward_total_composite_std": 0.29507479071617126, "reward_total_mean": 0.5821076035499573, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8833333253860474, "rewards/count_adherence/std": 0.077664315700531, "rewards/meter/mean": 0.6778547763824463, "rewards/meter/std": 0.2513851225376129, "rewards/total_composite/mean": 0.5821076035499573, "rewards/total_composite/std": 0.29507479071617126, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.039528250694275, "sampling/importance_sampling_ratio/min": 0.17914791405200958, "sampling/sampling_logp_difference/max": 1.71954345703125, "sampling/sampling_logp_difference/mean": 0.22136430442333221, "step": 74 }, { "clip_ratio/high_max": 0.08039004355669022, "clip_ratio/high_mean": 0.08039004355669022, "clip_ratio/low_mean": 0.11534719914197922, "clip_ratio/low_min": 0.11534719914197922, "clip_ratio/region_mean": 0.19573724269866943, "completions/clipped_ratio": 0.0, "completions/max_length": 154.0, "completions/max_terminated_length": 154.0, "completions/mean_length": 116.625, "completions/mean_terminated_length": 116.625, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 2.2147320955991745, "epoch": 0.0028962001853568117, "frac_reward_zero_std": 0.0, "grad_norm": 10.22510051727295, "learning_rate": 9.775757575757576e-06, "loss": 0.0601, "num_tokens": 152828.0, "reward": 0.43358057737350464, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.449047327041626, "reward_meter_std": 0.3430718183517456, "reward_std": 0.3434963822364807, "reward_total_composite_mean": 0.43358057737350464, "reward_total_composite_std": 0.3434963822364807, "reward_total_mean": 0.43358057737350464, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.449047327041626, "rewards/meter/std": 0.3430718183517456, "rewards/total_composite/mean": 0.43358057737350464, "rewards/total_composite/std": 0.3434963822364807, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.027030110359192, "sampling/importance_sampling_ratio/min": 0.1412452757358551, "sampling/sampling_logp_difference/max": 1.9572572708129883, "sampling/sampling_logp_difference/mean": 0.22043807804584503, "step": 75 }, { "clip_ratio/high_max": 0.12531227804720402, "clip_ratio/high_mean": 0.12531227804720402, "clip_ratio/low_mean": 0.09451080299913883, "clip_ratio/low_min": 0.09451080299913883, "clip_ratio/region_mean": 0.21982308104634285, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 55.25, "completions/mean_terminated_length": 55.25, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 3.0069815814495087, "epoch": 0.002934816187828236, "frac_reward_zero_std": 0.0, "grad_norm": 14.32852840423584, "learning_rate": 9.772727272727273e-06, "loss": 0.126, "num_tokens": 154686.0, "reward": 0.7078003883361816, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7078003883361816, "reward_meter_std": 0.4165555238723755, "reward_std": 0.4165555238723755, "reward_total_composite_mean": 0.7078003883361816, "reward_total_composite_std": 0.4165555238723755, "reward_total_mean": 0.7078003883361816, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7078003883361816, "rewards/meter/std": 0.4165555238723755, "rewards/total_composite/mean": 0.7078003883361816, "rewards/total_composite/std": 0.4165555238723755, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0568245649337769, "sampling/importance_sampling_ratio/min": 0.15894815325737, "sampling/sampling_logp_difference/max": 1.839177131652832, "sampling/sampling_logp_difference/mean": 0.22555997967720032, "step": 76 }, { "clip_ratio/high_max": 0.15934639982879162, "clip_ratio/high_mean": 0.15934639982879162, "clip_ratio/low_mean": 0.08010341972112656, "clip_ratio/low_min": 0.08010341972112656, "clip_ratio/region_mean": 0.23944981954991817, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 48.875, "completions/mean_terminated_length": 48.875, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 2.4677112847566605, "epoch": 0.0029734321902996604, "frac_reward_zero_std": 0.0, "grad_norm": 16.326507568359375, "learning_rate": 9.76969696969697e-06, "loss": 0.0662, "num_tokens": 156293.0, "reward": 0.6651355624198914, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6651355624198914, "reward_meter_std": 0.39970725774765015, "reward_std": 0.39970725774765015, "reward_total_composite_mean": 0.6651355624198914, "reward_total_composite_std": 0.39970725774765015, "reward_total_mean": 0.6651355624198914, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6651355624198914, "rewards/meter/std": 0.39970725774765015, "rewards/total_composite/mean": 0.6651355624198914, "rewards/total_composite/std": 0.39970725774765015, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.023947834968567, "sampling/importance_sampling_ratio/min": 0.20041508972644806, "sampling/sampling_logp_difference/max": 1.6073646545410156, "sampling/sampling_logp_difference/mean": 0.22144979238510132, "step": 77 }, { "clip_ratio/high_max": 0.06962481327354908, "clip_ratio/high_mean": 0.06962481327354908, "clip_ratio/low_mean": 0.1108522079885006, "clip_ratio/low_min": 0.1108522079885006, "clip_ratio/region_mean": 0.18047702126204967, "completions/clipped_ratio": 0.0, "completions/max_length": 197.0, "completions/max_terminated_length": 197.0, "completions/mean_length": 148.75, "completions/mean_terminated_length": 148.75, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 2.010555699467659, "epoch": 0.0030120481927710845, "frac_reward_zero_std": 0.0, "grad_norm": 8.249039649963379, "learning_rate": 9.766666666666667e-06, "loss": 0.073, "num_tokens": 159067.0, "reward": 0.35600364208221436, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.3600383400917053, "reward_meter_std": 0.2478063702583313, "reward_std": 0.2517344057559967, "reward_total_composite_mean": 0.35600364208221436, "reward_total_composite_std": 0.2517344057559967, "reward_total_mean": 0.35600364208221436, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.3600383400917053, "rewards/meter/std": 0.2478063702583313, "rewards/total_composite/mean": 0.35600364208221436, "rewards/total_composite/std": 0.2517344057559967, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0375880002975464, "sampling/importance_sampling_ratio/min": 0.1594703197479248, "sampling/sampling_logp_difference/max": 1.835897445678711, "sampling/sampling_logp_difference/mean": 0.20435652136802673, "step": 78 }, { "clip_ratio/high_max": 0.06808187626302242, "clip_ratio/high_mean": 0.06808187626302242, "clip_ratio/low_mean": 0.11126292496919632, "clip_ratio/low_min": 0.11126292496919632, "clip_ratio/region_mean": 0.17934480123221874, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 315.0, "completions/mean_terminated_length": 315.0, "completions/min_length": 211.0, "completions/min_terminated_length": 211.0, "entropy": 2.524102747440338, "epoch": 0.0030506641952425086, "frac_reward_zero_std": 0.0, "grad_norm": 5.221505165100098, "learning_rate": 9.763636363636365e-06, "loss": -0.0364, "num_tokens": 163339.0, "reward": 0.4219393730163574, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.10690447688102722, "reward_meter_mean": 0.4919300973415375, "reward_meter_std": 0.2810616195201874, "reward_std": 0.2564575970172882, "reward_total_composite_mean": 0.4219393730163574, "reward_total_composite_std": 0.2564575970172882, "reward_total_mean": 0.4219393730163574, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.10690447688102722, "rewards/meter/mean": 0.4919300973415375, "rewards/meter/std": 0.2810616195201874, "rewards/total_composite/mean": 0.4219393730163574, "rewards/total_composite/std": 0.2564575970172882, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0315507650375366, "sampling/importance_sampling_ratio/min": 0.13987872004508972, "sampling/sampling_logp_difference/max": 1.9669795036315918, "sampling/sampling_logp_difference/mean": 0.20352312922477722, "step": 79 }, { "clip_ratio/high_max": 0.1331654218956828, "clip_ratio/high_mean": 0.1331654218956828, "clip_ratio/low_mean": 0.05844542942941189, "clip_ratio/low_min": 0.05844542942941189, "clip_ratio/region_mean": 0.1916108513250947, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 55.25, "completions/mean_terminated_length": 55.25, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 2.6958227306604385, "epoch": 0.0030892801977139327, "frac_reward_zero_std": 0.0, "grad_norm": 15.532655715942383, "learning_rate": 9.760606060606062e-06, "loss": 0.0304, "num_tokens": 165053.0, "reward": 0.7070286273956299, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7070286273956299, "reward_meter_std": 0.3917587697505951, "reward_std": 0.3917587697505951, "reward_total_composite_mean": 0.7070286273956299, "reward_total_composite_std": 0.3917587697505951, "reward_total_mean": 0.7070286273956299, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7070286273956299, "rewards/meter/std": 0.3917587697505951, "rewards/total_composite/mean": 0.7070286273956299, "rewards/total_composite/std": 0.3917587697505951, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0330888032913208, "sampling/importance_sampling_ratio/min": 0.3348976671695709, "sampling/sampling_logp_difference/max": 1.0939302444458008, "sampling/sampling_logp_difference/mean": 0.2240315079689026, "step": 80 }, { "clip_ratio/high_max": 0.14205198176205158, "clip_ratio/high_mean": 0.14205198176205158, "clip_ratio/low_mean": 0.05491071380674839, "clip_ratio/low_min": 0.05491071380674839, "clip_ratio/region_mean": 0.19696269556879997, "completions/clipped_ratio": 0.0, "completions/max_length": 48.0, "completions/max_terminated_length": 48.0, "completions/mean_length": 40.875, "completions/mean_terminated_length": 40.875, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.8825555518269539, "epoch": 0.003127896200185357, "frac_reward_zero_std": 0.0, "grad_norm": 20.511795043945312, "learning_rate": 9.757575757575758e-06, "loss": -0.0475, "num_tokens": 166596.0, "reward": 0.6753016710281372, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6753016710281372, "reward_meter_std": 0.3897930681705475, "reward_std": 0.3897930681705475, "reward_total_composite_mean": 0.6753016710281372, "reward_total_composite_std": 0.3897930681705475, "reward_total_mean": 0.6753016710281372, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6753016710281372, "rewards/meter/std": 0.3897930681705475, "rewards/total_composite/mean": 0.6753016710281372, "rewards/total_composite/std": 0.3897930681705475, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9999099969863892, "sampling/importance_sampling_ratio/min": 0.11286209523677826, "sampling/sampling_logp_difference/max": 2.181588649749756, "sampling/sampling_logp_difference/mean": 0.17763756215572357, "step": 81 }, { "clip_ratio/high_max": 0.13331562653183937, "clip_ratio/high_mean": 0.13331562653183937, "clip_ratio/low_mean": 0.07035921700298786, "clip_ratio/low_min": 0.07035921700298786, "clip_ratio/region_mean": 0.20367484353482723, "completions/clipped_ratio": 0.0, "completions/max_length": 142.0, "completions/max_terminated_length": 142.0, "completions/mean_length": 74.75, "completions/mean_terminated_length": 74.75, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 2.4969915598630905, "epoch": 0.003166512202656781, "frac_reward_zero_std": 0.0, "grad_norm": 11.253849029541016, "learning_rate": 9.754545454545455e-06, "loss": 0.1511, "num_tokens": 168602.0, "reward": 0.6087316870689392, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7240955233573914, "reward_meter_std": 0.410367876291275, "reward_std": 0.4716724455356598, "reward_total_composite_mean": 0.6087316870689392, "reward_total_composite_std": 0.4716724455356598, "reward_total_mean": 0.6087316870689392, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7240955233573914, "rewards/meter/std": 0.410367876291275, "rewards/total_composite/mean": 0.6087316870689392, "rewards/total_composite/std": 0.4716724455356598, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0235908031463623, "sampling/importance_sampling_ratio/min": 0.17788389325141907, "sampling/sampling_logp_difference/max": 1.7266242504119873, "sampling/sampling_logp_difference/mean": 0.2147466540336609, "step": 82 }, { "clip_ratio/high_max": 0.10117011703550816, "clip_ratio/high_mean": 0.10117011703550816, "clip_ratio/low_mean": 0.09328041970729828, "clip_ratio/low_min": 0.09328041970729828, "clip_ratio/region_mean": 0.19445053674280643, "completions/clipped_ratio": 0.0, "completions/max_length": 264.0, "completions/max_terminated_length": 264.0, "completions/mean_length": 179.0, "completions/mean_terminated_length": 179.0, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 2.964230954647064, "epoch": 0.003205128205128205, "frac_reward_zero_std": 0.0, "grad_norm": 7.257972240447998, "learning_rate": 9.751515151515152e-06, "loss": -0.0125, "num_tokens": 171482.0, "reward": 0.4665340185165405, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_meter_mean": 0.5834420323371887, "reward_meter_std": 0.35952043533325195, "reward_std": 0.38863876461982727, "reward_total_composite_mean": 0.4665340185165405, "reward_total_composite_std": 0.38863879442214966, "reward_total_mean": 0.4665340185165405, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/meter/mean": 0.5834420323371887, "rewards/meter/std": 0.35952043533325195, "rewards/total_composite/mean": 0.4665340185165405, "rewards/total_composite/std": 0.38863879442214966, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.048352599143982, "sampling/importance_sampling_ratio/min": 0.12926006317138672, "sampling/sampling_logp_difference/max": 2.045928955078125, "sampling/sampling_logp_difference/mean": 0.20953215658664703, "step": 83 }, { "clip_ratio/high_max": 0.06479034759104252, "clip_ratio/high_mean": 0.06479034759104252, "clip_ratio/low_mean": 0.10067356191575527, "clip_ratio/low_min": 0.10067356191575527, "clip_ratio/region_mean": 0.1654639095067978, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 51.75, "completions/mean_terminated_length": 51.75, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 2.0957568138837814, "epoch": 0.003243744207599629, "frac_reward_zero_std": 0.0, "grad_norm": 20.633411407470703, "learning_rate": 9.74848484848485e-06, "loss": 0.1851, "num_tokens": 173072.0, "reward": 0.36928117275238037, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.36928117275238037, "reward_meter_std": 0.413564532995224, "reward_std": 0.413564532995224, "reward_total_composite_mean": 0.36928117275238037, "reward_total_composite_std": 0.413564532995224, "reward_total_mean": 0.36928117275238037, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.36928117275238037, "rewards/meter/std": 0.413564532995224, "rewards/total_composite/mean": 0.36928117275238037, "rewards/total_composite/std": 0.413564532995224, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0323539972305298, "sampling/importance_sampling_ratio/min": 0.176025852560997, "sampling/sampling_logp_difference/max": 1.7371244430541992, "sampling/sampling_logp_difference/mean": 0.20136907696723938, "step": 84 }, { "clip_ratio/high_max": 0.10693838447332382, "clip_ratio/high_mean": 0.10693838447332382, "clip_ratio/low_mean": 0.0852907095104456, "clip_ratio/low_min": 0.0852907095104456, "clip_ratio/region_mean": 0.19222909398376942, "completions/clipped_ratio": 0.0, "completions/max_length": 117.0, "completions/max_terminated_length": 117.0, "completions/mean_length": 90.125, "completions/mean_terminated_length": 90.125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 3.0091913044452667, "epoch": 0.0032823602100710537, "frac_reward_zero_std": 0.0, "grad_norm": 9.771239280700684, "learning_rate": 9.745454545454547e-06, "loss": -0.0541, "num_tokens": 175025.0, "reward": 0.6992344260215759, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6992344260215759, "reward_meter_std": 0.35877659916877747, "reward_std": 0.3587765693664551, "reward_total_composite_mean": 0.6992344260215759, "reward_total_composite_std": 0.35877659916877747, "reward_total_mean": 0.6992344260215759, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6992344260215759, "rewards/meter/std": 0.35877659916877747, "rewards/total_composite/mean": 0.6992344260215759, "rewards/total_composite/std": 0.35877659916877747, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0352109670639038, "sampling/importance_sampling_ratio/min": 0.15173391997814178, "sampling/sampling_logp_difference/max": 1.8856267929077148, "sampling/sampling_logp_difference/mean": 0.22157631814479828, "step": 85 }, { "clip_ratio/high_max": 0.05019771121442318, "clip_ratio/high_mean": 0.05019771121442318, "clip_ratio/low_mean": 0.13297666609287262, "clip_ratio/low_min": 0.13297666609287262, "clip_ratio/region_mean": 0.1831743773072958, "completions/clipped_ratio": 0.0, "completions/max_length": 254.0, "completions/max_terminated_length": 254.0, "completions/mean_length": 230.5, "completions/mean_terminated_length": 230.5, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "entropy": 2.359074369072914, "epoch": 0.0033209762125424778, "frac_reward_zero_std": 0.0, "grad_norm": 6.172459125518799, "learning_rate": 9.742424242424244e-06, "loss": 0.0244, "num_tokens": 178525.0, "reward": 0.2985811233520508, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_meter_mean": 0.2989215552806854, "reward_meter_std": 0.2961769700050354, "reward_std": 0.29654595255851746, "reward_total_composite_mean": 0.2985811233520508, "reward_total_composite_std": 0.29654595255851746, "reward_total_mean": 0.2985811233520508, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/meter/mean": 0.2989215552806854, "rewards/meter/std": 0.2961769700050354, "rewards/total_composite/mean": 0.2985811233520508, "rewards/total_composite/std": 0.29654595255851746, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0293720960617065, "sampling/importance_sampling_ratio/min": 0.09872201830148697, "sampling/sampling_logp_difference/max": 2.3154473304748535, "sampling/sampling_logp_difference/mean": 0.20698504149913788, "step": 86 }, { "clip_ratio/high_max": 0.09015538915991783, "clip_ratio/high_mean": 0.09015538915991783, "clip_ratio/low_mean": 0.07651033625006676, "clip_ratio/low_min": 0.07651033625006676, "clip_ratio/region_mean": 0.1666657254099846, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 45.5, "completions/mean_terminated_length": 45.5, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 1.9420476108789444, "epoch": 0.003359592215013902, "frac_reward_zero_std": 0.0, "grad_norm": 16.136564254760742, "learning_rate": 9.739393939393941e-06, "loss": 0.049, "num_tokens": 180121.0, "reward": 0.5413487553596497, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5413487553596497, "reward_meter_std": 0.41876327991485596, "reward_std": 0.41876330971717834, "reward_total_composite_mean": 0.5413487553596497, "reward_total_composite_std": 0.41876327991485596, "reward_total_mean": 0.5413487553596497, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5413487553596497, "rewards/meter/std": 0.41876327991485596, "rewards/total_composite/mean": 0.5413487553596497, "rewards/total_composite/std": 0.41876327991485596, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0214216709136963, "sampling/importance_sampling_ratio/min": 0.1677062213420868, "sampling/sampling_logp_difference/max": 1.7855415344238281, "sampling/sampling_logp_difference/mean": 0.20200753211975098, "step": 87 }, { "clip_ratio/high_max": 0.09816804714500904, "clip_ratio/high_mean": 0.09816804714500904, "clip_ratio/low_mean": 0.0701658520847559, "clip_ratio/low_min": 0.0701658520847559, "clip_ratio/region_mean": 0.16833389922976494, "completions/clipped_ratio": 0.0, "completions/max_length": 196.0, "completions/max_terminated_length": 196.0, "completions/mean_length": 171.75, "completions/mean_terminated_length": 171.75, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 1.721228078007698, "epoch": 0.003398208217485326, "frac_reward_zero_std": 0.0, "grad_norm": 7.712515830993652, "learning_rate": 9.736363636363637e-06, "loss": 0.0501, "num_tokens": 183183.0, "reward": 0.773547887802124, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.8209783434867859, "reward_meter_std": 0.20396688580513, "reward_std": 0.18800680339336395, "reward_total_composite_mean": 0.773547887802124, "reward_total_composite_std": 0.18800683319568634, "reward_total_mean": 0.773547887802124, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.8209783434867859, "rewards/meter/std": 0.20396688580513, "rewards/total_composite/mean": 0.773547887802124, "rewards/total_composite/std": 0.18800683319568634, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0262449979782104, "sampling/importance_sampling_ratio/min": 0.23252680897712708, "sampling/sampling_logp_difference/max": 1.458749771118164, "sampling/sampling_logp_difference/mean": 0.17306587100028992, "step": 88 }, { "clip_ratio/high_max": 0.12146328948438168, "clip_ratio/high_mean": 0.12146328948438168, "clip_ratio/low_mean": 0.07750886678695679, "clip_ratio/low_min": 0.07750886678695679, "clip_ratio/region_mean": 0.19897215627133846, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 88.625, "completions/mean_terminated_length": 88.625, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 2.467219904065132, "epoch": 0.00343682421995675, "frac_reward_zero_std": 0.0, "grad_norm": 11.22130298614502, "learning_rate": 9.733333333333334e-06, "loss": 0.0546, "num_tokens": 185260.0, "reward": 0.6381306648254395, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6381306648254395, "reward_meter_std": 0.44584599137306213, "reward_std": 0.44584596157073975, "reward_total_composite_mean": 0.6381306648254395, "reward_total_composite_std": 0.44584599137306213, "reward_total_mean": 0.6381306648254395, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6381306648254395, "rewards/meter/std": 0.44584599137306213, "rewards/total_composite/mean": 0.6381306648254395, "rewards/total_composite/std": 0.44584599137306213, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0421054363250732, "sampling/importance_sampling_ratio/min": 0.10935594886541367, "sampling/sampling_logp_difference/max": 2.2131471633911133, "sampling/sampling_logp_difference/mean": 0.19463911652565002, "step": 89 }, { "clip_ratio/high_max": 0.1579482052475214, "clip_ratio/high_mean": 0.1579482052475214, "clip_ratio/low_mean": 0.05007575824856758, "clip_ratio/low_min": 0.05007575824856758, "clip_ratio/region_mean": 0.20802396349608898, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 2.0693052262067795, "epoch": 0.003475440222428174, "frac_reward_zero_std": 0.0, "grad_norm": 12.855581283569336, "learning_rate": 9.730303030303031e-06, "loss": 0.0647, "num_tokens": 187088.0, "reward": 0.7145688533782959, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7145688533782959, "reward_meter_std": 0.3503323197364807, "reward_std": 0.3503322899341583, "reward_total_composite_mean": 0.7145688533782959, "reward_total_composite_std": 0.3503323197364807, "reward_total_mean": 0.7145688533782959, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7145688533782959, "rewards/meter/std": 0.3503323197364807, "rewards/total_composite/mean": 0.7145688533782959, "rewards/total_composite/std": 0.3503323197364807, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0224878787994385, "sampling/importance_sampling_ratio/min": 0.2688736319541931, "sampling/sampling_logp_difference/max": 1.3135137557983398, "sampling/sampling_logp_difference/mean": 0.18930701911449432, "step": 90 }, { "clip_ratio/high_max": 0.12906558997929096, "clip_ratio/high_mean": 0.12906558997929096, "clip_ratio/low_mean": 0.06466159597039223, "clip_ratio/low_min": 0.06466159597039223, "clip_ratio/region_mean": 0.1937271859496832, "completions/clipped_ratio": 0.0, "completions/max_length": 231.0, "completions/max_terminated_length": 231.0, "completions/mean_length": 192.125, "completions/mean_terminated_length": 192.125, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 3.0097380578517914, "epoch": 0.0035140562248995983, "frac_reward_zero_std": 0.0, "grad_norm": 6.48668909072876, "learning_rate": 9.727272727272728e-06, "loss": 0.1231, "num_tokens": 190233.0, "reward": 0.6788178086280823, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.9338054656982422, "reward_meter_std": 0.08479016274213791, "reward_std": 0.43503525853157043, "reward_total_composite_mean": 0.6788178086280823, "reward_total_composite_std": 0.43503525853157043, "reward_total_mean": 0.6788178086280823, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.9338054656982422, "rewards/meter/std": 0.08479016274213791, "rewards/total_composite/mean": 0.6788178086280823, "rewards/total_composite/std": 0.43503525853157043, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0515210628509521, "sampling/importance_sampling_ratio/min": 0.22454887628555298, "sampling/sampling_logp_difference/max": 1.493661880493164, "sampling/sampling_logp_difference/mean": 0.21253235638141632, "step": 91 }, { "clip_ratio/high_max": 0.11539113149046898, "clip_ratio/high_mean": 0.11539113149046898, "clip_ratio/low_mean": 0.13832124322652817, "clip_ratio/low_min": 0.13832124322652817, "clip_ratio/region_mean": 0.25371237471699715, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 54.125, "completions/mean_terminated_length": 54.125, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 1.927680492401123, "epoch": 0.0035526722273710224, "frac_reward_zero_std": 0.0, "grad_norm": 15.7993745803833, "learning_rate": 9.724242424242426e-06, "loss": -0.0808, "num_tokens": 191874.0, "reward": 0.532722532749176, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5675693154335022, "reward_meter_std": 0.4166634678840637, "reward_std": 0.4542304277420044, "reward_total_composite_mean": 0.532722532749176, "reward_total_composite_std": 0.4542304575443268, "reward_total_mean": 0.532722532749176, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5675693154335022, "rewards/meter/std": 0.4166634678840637, "rewards/total_composite/mean": 0.532722532749176, "rewards/total_composite/std": 0.4542304575443268, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004146695137024, "sampling/importance_sampling_ratio/min": 0.18035022914409637, "sampling/sampling_logp_difference/max": 1.7128546237945557, "sampling/sampling_logp_difference/mean": 0.22817404568195343, "step": 92 }, { "clip_ratio/high_max": 0.06774244643747807, "clip_ratio/high_mean": 0.06774244643747807, "clip_ratio/low_mean": 0.10987469553947449, "clip_ratio/low_min": 0.10987469553947449, "clip_ratio/region_mean": 0.17761714197695255, "completions/clipped_ratio": 0.0, "completions/max_length": 310.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 286.125, "completions/mean_terminated_length": 286.125, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 2.5551420152187347, "epoch": 0.0035912882298424465, "frac_reward_zero_std": 0.0, "grad_norm": 5.356238842010498, "learning_rate": 9.721212121212123e-06, "loss": 0.0068, "num_tokens": 196019.0, "reward": 0.3027617931365967, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.921875, "reward_count_adherence_std": 0.06469365209341049, "reward_meter_mean": 0.42547568678855896, "reward_meter_std": 0.28940528631210327, "reward_std": 0.2622292637825012, "reward_total_composite_mean": 0.3027617931365967, "reward_total_composite_std": 0.2622292935848236, "reward_total_mean": 0.3027617931365967, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.921875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/meter/mean": 0.42547568678855896, "rewards/meter/std": 0.28940528631210327, "rewards/total_composite/mean": 0.3027617931365967, "rewards/total_composite/std": 0.2622292935848236, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0348670482635498, "sampling/importance_sampling_ratio/min": 0.17745541036128998, "sampling/sampling_logp_difference/max": 1.7290358543395996, "sampling/sampling_logp_difference/mean": 0.19459845125675201, "step": 93 }, { "clip_ratio/high_max": 0.14526595920324326, "clip_ratio/high_mean": 0.14526595920324326, "clip_ratio/low_mean": 0.049512987956404686, "clip_ratio/low_min": 0.049512987956404686, "clip_ratio/region_mean": 0.19477894715964794, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 98.625, "completions/mean_terminated_length": 98.625, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 2.608750492334366, "epoch": 0.003629904232313871, "frac_reward_zero_std": 0.0, "grad_norm": 11.193428993225098, "learning_rate": 9.718181818181818e-06, "loss": -0.1159, "num_tokens": 198136.0, "reward": 0.6645107269287109, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8099383115768433, "reward_meter_std": 0.26826685667037964, "reward_std": 0.4238337576389313, "reward_total_composite_mean": 0.6645107269287109, "reward_total_composite_std": 0.4238337576389313, "reward_total_mean": 0.6645107269287109, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8099383115768433, "rewards/meter/std": 0.26826685667037964, "rewards/total_composite/mean": 0.6645107269287109, "rewards/total_composite/std": 0.4238337576389313, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0548158884048462, "sampling/importance_sampling_ratio/min": 0.1488121896982193, "sampling/sampling_logp_difference/max": 1.9050703048706055, "sampling/sampling_logp_difference/mean": 0.22230523824691772, "step": 94 }, { "clip_ratio/high_max": 0.047514619305729866, "clip_ratio/high_mean": 0.047514619305729866, "clip_ratio/low_mean": 0.02777777798473835, "clip_ratio/low_min": 0.02777777798473835, "clip_ratio/region_mean": 0.07529239729046822, "completions/clipped_ratio": 0.0, "completions/max_length": 27.0, "completions/max_terminated_length": 27.0, "completions/mean_length": 19.25, "completions/mean_terminated_length": 19.25, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 0.30344432033598423, "epoch": 0.003668520234785295, "frac_reward_zero_std": 0.0, "grad_norm": 55.388946533203125, "learning_rate": 9.715151515151516e-06, "loss": 0.1865, "num_tokens": 199458.0, "reward": 0.9113181829452515, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9113181829452515, "reward_meter_std": 0.23739001154899597, "reward_std": 0.23739001154899597, "reward_total_composite_mean": 0.9113181829452515, "reward_total_composite_std": 0.23739001154899597, "reward_total_mean": 0.9113181829452515, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9113181829452515, "rewards/meter/std": 0.23739001154899597, "rewards/total_composite/mean": 0.9113181829452515, "rewards/total_composite/std": 0.23739001154899597, "sampling/importance_sampling_ratio/max": 1.5570424795150757, "sampling/importance_sampling_ratio/mean": 0.9945163726806641, "sampling/importance_sampling_ratio/min": 0.36467787623405457, "sampling/sampling_logp_difference/max": 1.0087409019470215, "sampling/sampling_logp_difference/mean": 0.08319995552301407, "step": 95 }, { "clip_ratio/high_max": 0.12667790986597538, "clip_ratio/high_mean": 0.12667790986597538, "clip_ratio/low_mean": 0.07653119787573814, "clip_ratio/low_min": 0.07653119787573814, "clip_ratio/region_mean": 0.20320910774171352, "completions/clipped_ratio": 0.0, "completions/max_length": 136.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 94.5, "completions/mean_terminated_length": 94.5, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 3.304069072008133, "epoch": 0.0037071362372567192, "frac_reward_zero_std": 0.0, "grad_norm": 11.280510902404785, "learning_rate": 9.712121212121213e-06, "loss": 0.1775, "num_tokens": 201486.0, "reward": 0.6954616904258728, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6954616904258728, "reward_meter_std": 0.28914839029312134, "reward_std": 0.28914836049079895, "reward_total_composite_mean": 0.6954616904258728, "reward_total_composite_std": 0.28914839029312134, "reward_total_mean": 0.6954616904258728, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6954616904258728, "rewards/meter/std": 0.28914839029312134, "rewards/total_composite/mean": 0.6954616904258728, "rewards/total_composite/std": 0.28914839029312134, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.037481665611267, "sampling/importance_sampling_ratio/min": 0.13818837702274323, "sampling/sampling_logp_difference/max": 1.9791374206542969, "sampling/sampling_logp_difference/mean": 0.2430281937122345, "step": 96 }, { "clip_ratio/high_max": 0.11797327920794487, "clip_ratio/high_mean": 0.11797327920794487, "clip_ratio/low_mean": 0.10330966301262379, "clip_ratio/low_min": 0.10330966301262379, "clip_ratio/region_mean": 0.22128294222056866, "completions/clipped_ratio": 0.0, "completions/max_length": 92.0, "completions/max_terminated_length": 92.0, "completions/mean_length": 77.75, "completions/mean_terminated_length": 77.75, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 2.470811814069748, "epoch": 0.0037457522397281433, "frac_reward_zero_std": 0.0, "grad_norm": 11.798491477966309, "learning_rate": 9.70909090909091e-06, "loss": 0.046, "num_tokens": 203468.0, "reward": 0.44233888387680054, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.44233888387680054, "reward_meter_std": 0.3354645371437073, "reward_std": 0.3354645073413849, "reward_total_composite_mean": 0.44233888387680054, "reward_total_composite_std": 0.3354645371437073, "reward_total_mean": 0.44233888387680054, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.44233888387680054, "rewards/meter/std": 0.3354645371437073, "rewards/total_composite/mean": 0.44233888387680054, "rewards/total_composite/std": 0.3354645371437073, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0586459636688232, "sampling/importance_sampling_ratio/min": 0.20629987120628357, "sampling/sampling_logp_difference/max": 1.5784244537353516, "sampling/sampling_logp_difference/mean": 0.23818424344062805, "step": 97 }, { "clip_ratio/high_max": 0.08083950355648994, "clip_ratio/high_mean": 0.08083950355648994, "clip_ratio/low_mean": 0.12072680331766605, "clip_ratio/low_min": 0.12072680331766605, "clip_ratio/region_mean": 0.201566306874156, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 79.5, "completions/mean_terminated_length": 79.5, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 2.8207843005657196, "epoch": 0.0037843682421995675, "frac_reward_zero_std": 0.0, "grad_norm": 10.285151481628418, "learning_rate": 9.706060606060606e-06, "loss": -0.0341, "num_tokens": 205368.0, "reward": 0.364272803068161, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.2357022762298584, "reward_meter_mean": 0.458587646484375, "reward_meter_std": 0.4029514789581299, "reward_std": 0.41265976428985596, "reward_total_composite_mean": 0.364272803068161, "reward_total_composite_std": 0.41265976428985596, "reward_total_mean": 0.364272803068161, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.2357022762298584, "rewards/meter/mean": 0.458587646484375, "rewards/meter/std": 0.4029514789581299, "rewards/total_composite/mean": 0.364272803068161, "rewards/total_composite/std": 0.41265976428985596, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0597635507583618, "sampling/importance_sampling_ratio/min": 0.22730112075805664, "sampling/sampling_logp_difference/max": 1.4814796447753906, "sampling/sampling_logp_difference/mean": 0.22706757485866547, "step": 98 }, { "clip_ratio/high_max": 0.06744235754013062, "clip_ratio/high_mean": 0.06744235754013062, "clip_ratio/low_mean": 0.062378629110753536, "clip_ratio/low_min": 0.062378629110753536, "clip_ratio/region_mean": 0.12982098665088415, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 89.125, "completions/mean_terminated_length": 89.125, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 1.236207775771618, "epoch": 0.0038229842446709916, "frac_reward_zero_std": 0.0, "grad_norm": 11.669129371643066, "learning_rate": 9.703030303030305e-06, "loss": 0.0857, "num_tokens": 207481.0, "reward": 0.796177864074707, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.796177864074707, "reward_meter_std": 0.30666306614875793, "reward_std": 0.30666306614875793, "reward_total_composite_mean": 0.796177864074707, "reward_total_composite_std": 0.30666306614875793, "reward_total_mean": 0.796177864074707, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.796177864074707, "rewards/meter/std": 0.30666306614875793, "rewards/total_composite/mean": 0.796177864074707, "rewards/total_composite/std": 0.30666306614875793, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0176408290863037, "sampling/importance_sampling_ratio/min": 0.21204693615436554, "sampling/sampling_logp_difference/max": 1.550947666168213, "sampling/sampling_logp_difference/mean": 0.16059812903404236, "step": 99 }, { "clip_ratio/high_max": 0.10258511640131474, "clip_ratio/high_mean": 0.10258511640131474, "clip_ratio/low_mean": 0.0948890820145607, "clip_ratio/low_min": 0.0948890820145607, "clip_ratio/region_mean": 0.19747419841587543, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 51.125, "completions/mean_terminated_length": 51.125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 2.0875614881515503, "epoch": 0.0038616002471424157, "frac_reward_zero_std": 0.0, "grad_norm": 16.21261978149414, "learning_rate": 9.7e-06, "loss": 0.0274, "num_tokens": 209138.0, "reward": 0.4757336378097534, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4757336378097534, "reward_meter_std": 0.39768749475479126, "reward_std": 0.39768749475479126, "reward_total_composite_mean": 0.4757336378097534, "reward_total_composite_std": 0.39768749475479126, "reward_total_mean": 0.4757336378097534, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4757336378097534, "rewards/meter/std": 0.39768749475479126, "rewards/total_composite/mean": 0.4757336378097534, "rewards/total_composite/std": 0.39768749475479126, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.024910569190979, "sampling/importance_sampling_ratio/min": 0.2545141875743866, "sampling/sampling_logp_difference/max": 1.368398666381836, "sampling/sampling_logp_difference/mean": 0.2088787704706192, "step": 100 }, { "epoch": 0.0038616002471424157, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.038461538461538464, "eval_completions/max_length": 412.9230769230769, "eval_completions/max_terminated_length": 374.53846153846155, "eval_completions/mean_length": 189.1153846153846, "eval_completions/mean_terminated_length": 176.4835181603065, "eval_completions/min_length": 39.15384615384615, "eval_completions/min_terminated_length": 39.15384615384615, "eval_entropy": 2.585876281444843, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 209138.0, "eval_reward": 0.36039777329334843, "eval_reward_arabic_clean_mean": 0.9519230769230769, "eval_reward_arabic_clean_std": 0.13598207097787124, "eval_reward_count_adherence_mean": 0.9531642427811255, "eval_reward_count_adherence_std": 0.06681363327571979, "eval_reward_meter_mean": 0.38446200467073, "eval_reward_meter_std": 0.3474533214018895, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.36039777329334843, "eval_reward_total_composite_std": 0.34838862602527326, "eval_reward_total_mean": 0.36039777329334843, "eval_rewards/arabic_clean/mean": 0.9519230769230769, "eval_rewards/arabic_clean/std": 0.13598207097787124, "eval_rewards/count_adherence/mean": 0.9531642427811255, "eval_rewards/count_adherence/std": 0.06681363327571979, "eval_rewards/meter/mean": 0.38446200467073, "eval_rewards/meter/std": 0.3474533214018895, "eval_rewards/total_composite/mean": 0.36039777329334843, "eval_rewards/total_composite/std": 0.34838862602527326, "eval_runtime": 76.843, "eval_samples_per_second": 1.353, "eval_sampling/importance_sampling_ratio/max": 1.6659116469896758, "eval_sampling/importance_sampling_ratio/mean": 1.043038276525644, "eval_sampling/importance_sampling_ratio/min": 0.2675319703725668, "eval_sampling/sampling_logp_difference/max": 1.3432146219106822, "eval_sampling/sampling_logp_difference/mean": 0.1495220626776035, "eval_steps_per_second": 0.169, "step": 100 }, { "clip_ratio/high_max": 0.06029390636831522, "clip_ratio/high_mean": 0.06029390636831522, "clip_ratio/low_mean": 0.10309341456741095, "clip_ratio/low_min": 0.10309341456741095, "clip_ratio/region_mean": 0.16338732093572617, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 64.875, "completions/mean_terminated_length": 64.875, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 2.11554679274559, "epoch": 0.0039002162496138398, "frac_reward_zero_std": 0.0, "grad_norm": 13.529213905334473, "learning_rate": 9.696969696969698e-06, "loss": -0.0129, "num_tokens": 210953.0, "reward": 0.3698737621307373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_meter_mean": 0.40993648767471313, "reward_meter_std": 0.29322153329849243, "reward_std": 0.26750627160072327, "reward_total_composite_mean": 0.3698737621307373, "reward_total_composite_std": 0.26750627160072327, "reward_total_mean": 0.3698737621307373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/meter/mean": 0.40993648767471313, "rewards/meter/std": 0.29322153329849243, "rewards/total_composite/mean": 0.3698737621307373, "rewards/total_composite/std": 0.26750627160072327, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0172652006149292, "sampling/importance_sampling_ratio/min": 0.2798984944820404, "sampling/sampling_logp_difference/max": 1.2733283042907715, "sampling/sampling_logp_difference/mean": 0.203210711479187, "step": 101 }, { "clip_ratio/high_max": 0.10505400598049164, "clip_ratio/high_mean": 0.10505400598049164, "clip_ratio/low_mean": 0.09572393447160721, "clip_ratio/low_min": 0.09572393447160721, "clip_ratio/region_mean": 0.20077794045209885, "completions/clipped_ratio": 0.0, "completions/max_length": 139.0, "completions/max_terminated_length": 139.0, "completions/mean_length": 109.125, "completions/mean_terminated_length": 109.125, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 2.91497141122818, "epoch": 0.003938832252085264, "frac_reward_zero_std": 0.0, "grad_norm": 9.461191177368164, "learning_rate": 9.693939393939395e-06, "loss": -0.0229, "num_tokens": 213210.0, "reward": 0.3811372220516205, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5020676851272583, "reward_meter_std": 0.33497413992881775, "reward_std": 0.3171204626560211, "reward_total_composite_mean": 0.3811372220516205, "reward_total_composite_std": 0.3171204626560211, "reward_total_mean": 0.3811372220516205, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5020676851272583, "rewards/meter/std": 0.33497413992881775, "rewards/total_composite/mean": 0.3811372220516205, "rewards/total_composite/std": 0.3171204626560211, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0544703006744385, "sampling/importance_sampling_ratio/min": 0.2506718039512634, "sampling/sampling_logp_difference/max": 1.383610725402832, "sampling/sampling_logp_difference/mean": 0.2128397673368454, "step": 102 }, { "clip_ratio/high_max": 0.08475394546985626, "clip_ratio/high_mean": 0.08475394546985626, "clip_ratio/low_mean": 0.10796530079096556, "clip_ratio/low_min": 0.10796530079096556, "clip_ratio/region_mean": 0.19271924626082182, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 51.75, "completions/mean_terminated_length": 51.75, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 2.3894866704940796, "epoch": 0.003977448254556688, "frac_reward_zero_std": 0.0, "grad_norm": 13.193228721618652, "learning_rate": 9.690909090909092e-06, "loss": 0.0772, "num_tokens": 215040.0, "reward": 0.48240411281585693, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.48240411281585693, "reward_meter_std": 0.4074939489364624, "reward_std": 0.4074939489364624, "reward_total_composite_mean": 0.48240411281585693, "reward_total_composite_std": 0.4074939489364624, "reward_total_mean": 0.48240411281585693, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.48240411281585693, "rewards/meter/std": 0.4074939489364624, "rewards/total_composite/mean": 0.48240411281585693, "rewards/total_composite/std": 0.4074939489364624, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0494117736816406, "sampling/importance_sampling_ratio/min": 0.19952771067619324, "sampling/sampling_logp_difference/max": 1.611802101135254, "sampling/sampling_logp_difference/mean": 0.217100128531456, "step": 103 }, { "clip_ratio/high_max": 0.1372815314680338, "clip_ratio/high_mean": 0.1372815314680338, "clip_ratio/low_mean": 0.07232538796961308, "clip_ratio/low_min": 0.07232538796961308, "clip_ratio/region_mean": 0.20960691943764687, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 46.5, "completions/mean_terminated_length": 46.5, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 2.75657719373703, "epoch": 0.004016064257028112, "frac_reward_zero_std": 0.0, "grad_norm": 14.126129150390625, "learning_rate": 9.687878787878788e-06, "loss": -0.0086, "num_tokens": 216684.0, "reward": 0.6793533563613892, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6793533563613892, "reward_meter_std": 0.37877920269966125, "reward_std": 0.37877923250198364, "reward_total_composite_mean": 0.6793533563613892, "reward_total_composite_std": 0.37877920269966125, "reward_total_mean": 0.6793533563613892, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6793533563613892, "rewards/meter/std": 0.37877920269966125, "rewards/total_composite/mean": 0.6793533563613892, "rewards/total_composite/std": 0.37877920269966125, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0407179594039917, "sampling/importance_sampling_ratio/min": 0.22745074331760406, "sampling/sampling_logp_difference/max": 1.4808216094970703, "sampling/sampling_logp_difference/mean": 0.22054526209831238, "step": 104 }, { "clip_ratio/high_max": 0.06583333387970924, "clip_ratio/high_mean": 0.06583333387970924, "clip_ratio/low_mean": 0.144812298938632, "clip_ratio/low_min": 0.144812298938632, "clip_ratio/region_mean": 0.21064563281834126, "completions/clipped_ratio": 0.0, "completions/max_length": 51.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 44.125, "completions/mean_terminated_length": 44.125, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 1.7920314818620682, "epoch": 0.004054680259499536, "frac_reward_zero_std": 0.0, "grad_norm": 16.32758140563965, "learning_rate": 9.684848484848487e-06, "loss": 0.0212, "num_tokens": 218277.0, "reward": 0.4026336967945099, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4026336967945099, "reward_meter_std": 0.35716599225997925, "reward_std": 0.35716599225997925, "reward_total_composite_mean": 0.4026336967945099, "reward_total_composite_std": 0.35716599225997925, "reward_total_mean": 0.4026336967945099, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4026336967945099, "rewards/meter/std": 0.35716599225997925, "rewards/total_composite/mean": 0.4026336967945099, "rewards/total_composite/std": 0.35716599225997925, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0534363985061646, "sampling/importance_sampling_ratio/min": 0.2275705188512802, "sampling/sampling_logp_difference/max": 1.480295181274414, "sampling/sampling_logp_difference/mean": 0.16692768037319183, "step": 105 }, { "clip_ratio/high_max": 0.10586144216358662, "clip_ratio/high_mean": 0.10586144216358662, "clip_ratio/low_mean": 0.13967670314013958, "clip_ratio/low_min": 0.13967670314013958, "clip_ratio/region_mean": 0.2455381453037262, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 76.875, "completions/mean_terminated_length": 76.875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 3.015815883874893, "epoch": 0.004093296261970961, "frac_reward_zero_std": 0.0, "grad_norm": 13.200583457946777, "learning_rate": 9.681818181818182e-06, "loss": -0.0234, "num_tokens": 220292.0, "reward": 0.3593082129955292, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.3820000886917114, "reward_meter_std": 0.3840147852897644, "reward_std": 0.3966972529888153, "reward_total_composite_mean": 0.3593082129955292, "reward_total_composite_std": 0.3966972827911377, "reward_total_mean": 0.3593082129955292, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.3820000886917114, "rewards/meter/std": 0.3840147852897644, "rewards/total_composite/mean": 0.3593082129955292, "rewards/total_composite/std": 0.3966972827911377, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0527663230895996, "sampling/importance_sampling_ratio/min": 0.21993055939674377, "sampling/sampling_logp_difference/max": 1.5144433975219727, "sampling/sampling_logp_difference/mean": 0.24001426994800568, "step": 106 }, { "clip_ratio/high_max": 0.1405453272163868, "clip_ratio/high_mean": 0.1405453272163868, "clip_ratio/low_mean": 0.06378891877830029, "clip_ratio/low_min": 0.06378891877830029, "clip_ratio/region_mean": 0.20433424599468708, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 270.875, "completions/mean_terminated_length": 270.875, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 3.4069304764270782, "epoch": 0.004131912264442385, "frac_reward_zero_std": 0.0, "grad_norm": 5.8178300857543945, "learning_rate": 9.67878787878788e-06, "loss": 0.0184, "num_tokens": 223987.0, "reward": 0.5568721294403076, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_meter_mean": 0.6609253883361816, "reward_meter_std": 0.2803538143634796, "reward_std": 0.3528820872306824, "reward_total_composite_mean": 0.5568721294403076, "reward_total_composite_std": 0.3528820872306824, "reward_total_mean": 0.5568721294403076, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/meter/mean": 0.6609253883361816, "rewards/meter/std": 0.2803538143634796, "rewards/total_composite/mean": 0.5568721294403076, "rewards/total_composite/std": 0.3528820872306824, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0432038307189941, "sampling/importance_sampling_ratio/min": 0.2048678696155548, "sampling/sampling_logp_difference/max": 1.5853900909423828, "sampling/sampling_logp_difference/mean": 0.22177748382091522, "step": 107 }, { "clip_ratio/high_max": 0.10650285705924034, "clip_ratio/high_mean": 0.10650285705924034, "clip_ratio/low_mean": 0.08765609562397003, "clip_ratio/low_min": 0.08765609562397003, "clip_ratio/region_mean": 0.19415895268321037, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 49.375, "completions/mean_terminated_length": 49.375, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 2.991265833377838, "epoch": 0.004170528266913809, "frac_reward_zero_std": 0.0, "grad_norm": 16.24658203125, "learning_rate": 9.675757575757577e-06, "loss": 0.0163, "num_tokens": 225574.0, "reward": 0.4669821858406067, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4785667061805725, "reward_meter_std": 0.40862172842025757, "reward_std": 0.4222123324871063, "reward_total_composite_mean": 0.4669821858406067, "reward_total_composite_std": 0.4222123324871063, "reward_total_mean": 0.4669821858406067, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4785667061805725, "rewards/meter/std": 0.40862172842025757, "rewards/total_composite/mean": 0.4669821858406067, "rewards/total_composite/std": 0.4222123324871063, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0536181926727295, "sampling/importance_sampling_ratio/min": 0.2530842125415802, "sampling/sampling_logp_difference/max": 1.374032974243164, "sampling/sampling_logp_difference/mean": 0.22205865383148193, "step": 108 }, { "clip_ratio/high_max": 0.09419741854071617, "clip_ratio/high_mean": 0.09419741854071617, "clip_ratio/low_mean": 0.07784751430153847, "clip_ratio/low_min": 0.07784751430153847, "clip_ratio/region_mean": 0.17204493284225464, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 97.0, "completions/mean_terminated_length": 97.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 3.290193408727646, "epoch": 0.0042091442693852335, "frac_reward_zero_std": 0.0, "grad_norm": 10.769991874694824, "learning_rate": 9.672727272727274e-06, "loss": -0.0954, "num_tokens": 227646.0, "reward": 0.525036096572876, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.525036096572876, "reward_meter_std": 0.32685554027557373, "reward_std": 0.32685554027557373, "reward_total_composite_mean": 0.525036096572876, "reward_total_composite_std": 0.32685554027557373, "reward_total_mean": 0.525036096572876, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.525036096572876, "rewards/meter/std": 0.32685554027557373, "rewards/total_composite/mean": 0.525036096572876, "rewards/total_composite/std": 0.32685554027557373, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.024579405784607, "sampling/importance_sampling_ratio/min": 0.292019248008728, "sampling/sampling_logp_difference/max": 1.2309355735778809, "sampling/sampling_logp_difference/mean": 0.230497807264328, "step": 109 }, { "clip_ratio/high_max": 0.11844915337860584, "clip_ratio/high_mean": 0.11844915337860584, "clip_ratio/low_mean": 0.0400856789201498, "clip_ratio/low_min": 0.0400856789201498, "clip_ratio/region_mean": 0.15853483229875565, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 62.375, "completions/mean_terminated_length": 62.375, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 2.1576623022556305, "epoch": 0.004247760271856658, "frac_reward_zero_std": 0.0, "grad_norm": 13.498187065124512, "learning_rate": 9.66969696969697e-06, "loss": -0.0629, "num_tokens": 229401.0, "reward": 0.7822655439376831, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7822750210762024, "reward_meter_std": 0.3426264226436615, "reward_std": 0.3426511585712433, "reward_total_composite_mean": 0.7822655439376831, "reward_total_composite_std": 0.3426511585712433, "reward_total_mean": 0.7822655439376831, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7822750210762024, "rewards/meter/std": 0.3426264226436615, "rewards/total_composite/mean": 0.7822655439376831, "rewards/total_composite/std": 0.3426511585712433, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0487669706344604, "sampling/importance_sampling_ratio/min": 0.3012003004550934, "sampling/sampling_logp_difference/max": 1.3505010604858398, "sampling/sampling_logp_difference/mean": 0.20653371512889862, "step": 110 }, { "clip_ratio/high_max": 0.10812411084771156, "clip_ratio/high_mean": 0.10812411084771156, "clip_ratio/low_mean": 0.07732126116752625, "clip_ratio/low_min": 0.07732126116752625, "clip_ratio/region_mean": 0.1854453720152378, "completions/clipped_ratio": 0.0, "completions/max_length": 136.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 112.625, "completions/mean_terminated_length": 112.625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 2.363381966948509, "epoch": 0.004286376274328082, "frac_reward_zero_std": 0.0, "grad_norm": 9.501466751098633, "learning_rate": 9.666666666666667e-06, "loss": 0.0631, "num_tokens": 231622.0, "reward": 0.5836721062660217, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.6113415956497192, "reward_meter_std": 0.39460012316703796, "reward_std": 0.3801315128803253, "reward_total_composite_mean": 0.5836721062660217, "reward_total_composite_std": 0.3801315426826477, "reward_total_mean": 0.5836721062660217, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.6113415956497192, "rewards/meter/std": 0.39460012316703796, "rewards/total_composite/mean": 0.5836721062660217, "rewards/total_composite/std": 0.3801315426826477, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0375341176986694, "sampling/importance_sampling_ratio/min": 0.22895215451717377, "sampling/sampling_logp_difference/max": 1.4742422103881836, "sampling/sampling_logp_difference/mean": 0.1836644411087036, "step": 111 }, { "clip_ratio/high_max": 0.03674242552369833, "clip_ratio/high_mean": 0.03674242552369833, "clip_ratio/low_mean": 0.10191993555054069, "clip_ratio/low_min": 0.10191993555054069, "clip_ratio/region_mean": 0.13866236107423902, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 26.5, "completions/mean_terminated_length": 26.5, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 1.8556535243988037, "epoch": 0.004324992276799506, "frac_reward_zero_std": 0.0, "grad_norm": 21.516904830932617, "learning_rate": 9.663636363636364e-06, "loss": 0.1002, "num_tokens": 233106.0, "reward": 0.30022066831588745, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.30022066831588745, "reward_meter_std": 0.3768041133880615, "reward_std": 0.3768041133880615, "reward_total_composite_mean": 0.30022066831588745, "reward_total_composite_std": 0.3768041133880615, "reward_total_mean": 0.30022066831588745, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.30022066831588745, "rewards/meter/std": 0.3768041133880615, "rewards/total_composite/mean": 0.30022066831588745, "rewards/total_composite/std": 0.3768041133880615, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.041845679283142, "sampling/importance_sampling_ratio/min": 0.3161206543445587, "sampling/sampling_logp_difference/max": 1.1516313552856445, "sampling/sampling_logp_difference/mean": 0.20119836926460266, "step": 112 }, { "clip_ratio/high_max": 0.12023117486387491, "clip_ratio/high_mean": 0.12023117486387491, "clip_ratio/low_mean": 0.0982854887843132, "clip_ratio/low_min": 0.0982854887843132, "clip_ratio/region_mean": 0.21851666364818811, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 54.5, "completions/mean_terminated_length": 54.5, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 2.915360987186432, "epoch": 0.00436360827927093, "frac_reward_zero_std": 0.0, "grad_norm": 13.721908569335938, "learning_rate": 9.660606060606061e-06, "loss": -0.0059, "num_tokens": 234710.0, "reward": 0.543341875076294, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.543341875076294, "reward_meter_std": 0.3554902970790863, "reward_std": 0.3554903268814087, "reward_total_composite_mean": 0.543341875076294, "reward_total_composite_std": 0.3554902970790863, "reward_total_mean": 0.543341875076294, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.543341875076294, "rewards/meter/std": 0.3554902970790863, "rewards/total_composite/mean": 0.543341875076294, "rewards/total_composite/std": 0.3554902970790863, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0506563186645508, "sampling/importance_sampling_ratio/min": 0.3279053270816803, "sampling/sampling_logp_difference/max": 1.115030288696289, "sampling/sampling_logp_difference/mean": 0.22269107401371002, "step": 113 }, { "clip_ratio/high_max": 0.0366877019405365, "clip_ratio/high_mean": 0.0366877019405365, "clip_ratio/low_mean": 0.11057115998119116, "clip_ratio/low_min": 0.11057115998119116, "clip_ratio/region_mean": 0.14725886192172766, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 85.875, "completions/mean_terminated_length": 85.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 2.166059359908104, "epoch": 0.004402224281742354, "frac_reward_zero_std": 0.0, "grad_norm": 10.23147964477539, "learning_rate": 9.657575757575758e-06, "loss": -0.0029, "num_tokens": 236805.0, "reward": 0.21646319329738617, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.21813370287418365, "reward_meter_std": 0.3496148884296417, "reward_std": 0.35061758756637573, "reward_total_composite_mean": 0.21646319329738617, "reward_total_composite_std": 0.35061758756637573, "reward_total_mean": 0.21646319329738617, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.21813370287418365, "rewards/meter/std": 0.3496148884296417, "rewards/total_composite/mean": 0.21646319329738617, "rewards/total_composite/std": 0.35061758756637573, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0350022315979004, "sampling/importance_sampling_ratio/min": 0.295941025018692, "sampling/sampling_logp_difference/max": 1.217595100402832, "sampling/sampling_logp_difference/mean": 0.19407296180725098, "step": 114 }, { "clip_ratio/high_max": 0.08701479807496071, "clip_ratio/high_mean": 0.08701479807496071, "clip_ratio/low_mean": 0.11949973180890083, "clip_ratio/low_min": 0.11949973180890083, "clip_ratio/region_mean": 0.20651452988386154, "completions/clipped_ratio": 0.0, "completions/max_length": 136.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 117.75, "completions/mean_terminated_length": 117.75, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 2.1785334795713425, "epoch": 0.004440840284213778, "frac_reward_zero_std": 0.0, "grad_norm": 9.909034729003906, "learning_rate": 9.654545454545456e-06, "loss": 0.0735, "num_tokens": 239099.0, "reward": 0.4193575978279114, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4193575978279114, "reward_meter_std": 0.3706149458885193, "reward_std": 0.3706149160861969, "reward_total_composite_mean": 0.4193575978279114, "reward_total_composite_std": 0.3706149458885193, "reward_total_mean": 0.4193575978279114, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4193575978279114, "rewards/meter/std": 0.3706149458885193, "rewards/total_composite/mean": 0.4193575978279114, "rewards/total_composite/std": 0.3706149458885193, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.024393916130066, "sampling/importance_sampling_ratio/min": 0.13901562988758087, "sampling/sampling_logp_difference/max": 1.9731688499450684, "sampling/sampling_logp_difference/mean": 0.21766838431358337, "step": 115 }, { "clip_ratio/high_max": 0.15062034130096436, "clip_ratio/high_mean": 0.15062034130096436, "clip_ratio/low_mean": 0.05248366016894579, "clip_ratio/low_min": 0.05248366016894579, "clip_ratio/region_mean": 0.20310400146991014, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 56.125, "completions/mean_terminated_length": 56.125, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 2.4604494273662567, "epoch": 0.004479456286685202, "frac_reward_zero_std": 0.0, "grad_norm": 13.37387752532959, "learning_rate": 9.651515151515153e-06, "loss": 0.0996, "num_tokens": 240788.0, "reward": 0.6912834644317627, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6912834644317627, "reward_meter_std": 0.39879509806632996, "reward_std": 0.39879506826400757, "reward_total_composite_mean": 0.6912834644317627, "reward_total_composite_std": 0.39879509806632996, "reward_total_mean": 0.6912834644317627, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6912834644317627, "rewards/meter/std": 0.39879509806632996, "rewards/total_composite/mean": 0.6912834644317627, "rewards/total_composite/std": 0.39879509806632996, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0366685390472412, "sampling/importance_sampling_ratio/min": 0.216609388589859, "sampling/sampling_logp_difference/max": 1.529659628868103, "sampling/sampling_logp_difference/mean": 0.18541668355464935, "step": 116 }, { "clip_ratio/high_max": 0.0818505771458149, "clip_ratio/high_mean": 0.0818505771458149, "clip_ratio/low_mean": 0.15885628014802933, "clip_ratio/low_min": 0.15885628014802933, "clip_ratio/region_mean": 0.24070685729384422, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 24.875, "completions/mean_terminated_length": 24.875, "completions/min_length": 16.0, "completions/min_terminated_length": 16.0, "entropy": 2.409518927335739, "epoch": 0.004518072289156626, "frac_reward_zero_std": 0.0, "grad_norm": 22.580102920532227, "learning_rate": 9.648484848484849e-06, "loss": 0.1027, "num_tokens": 242291.0, "reward": 0.4115869998931885, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4409791827201843, "reward_meter_std": 0.46491894125938416, "reward_std": 0.48671311140060425, "reward_total_composite_mean": 0.4115869998931885, "reward_total_composite_std": 0.48671314120292664, "reward_total_mean": 0.4115869998931885, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4409791827201843, "rewards/meter/std": 0.46491894125938416, "rewards/total_composite/mean": 0.4115869998931885, "rewards/total_composite/std": 0.48671314120292664, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0259357690811157, "sampling/importance_sampling_ratio/min": 0.20900969207286835, "sampling/sampling_logp_difference/max": 1.5653746128082275, "sampling/sampling_logp_difference/mean": 0.2450607866048813, "step": 117 }, { "clip_ratio/high_max": 0.14602509513497353, "clip_ratio/high_mean": 0.14602509513497353, "clip_ratio/low_mean": 0.0561376241967082, "clip_ratio/low_min": 0.0561376241967082, "clip_ratio/region_mean": 0.20216271933168173, "completions/clipped_ratio": 0.0, "completions/max_length": 180.0, "completions/max_terminated_length": 180.0, "completions/mean_length": 159.5, "completions/mean_terminated_length": 159.5, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 2.5699357092380524, "epoch": 0.00455668829162805, "frac_reward_zero_std": 0.0, "grad_norm": 8.194822311401367, "learning_rate": 9.645454545454548e-06, "loss": 0.0803, "num_tokens": 244935.0, "reward": 0.6134665012359619, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.6182658672332764, "reward_meter_std": 0.35096442699432373, "reward_std": 0.3578221797943115, "reward_total_composite_mean": 0.6134665012359619, "reward_total_composite_std": 0.3578221797943115, "reward_total_mean": 0.6134665012359619, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.6182658672332764, "rewards/meter/std": 0.35096442699432373, "rewards/total_composite/mean": 0.6134665012359619, "rewards/total_composite/std": 0.3578221797943115, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.02511465549469, "sampling/importance_sampling_ratio/min": 0.1597553789615631, "sampling/sampling_logp_difference/max": 1.8341115713119507, "sampling/sampling_logp_difference/mean": 0.20374369621276855, "step": 118 }, { "clip_ratio/high_max": 0.18722888734191656, "clip_ratio/high_mean": 0.18722888734191656, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/region_mean": 0.19972888752818108, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 29.125, "completions/mean_terminated_length": 29.125, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 2.4730520844459534, "epoch": 0.0045953042940994745, "frac_reward_zero_std": 0.0, "grad_norm": 15.294441223144531, "learning_rate": 9.642424242424243e-06, "loss": -0.0893, "num_tokens": 246392.0, "reward": 0.8526638150215149, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9771455526351929, "reward_meter_std": 0.02793486975133419, "reward_std": 0.3455761969089508, "reward_total_composite_mean": 0.8526638150215149, "reward_total_composite_std": 0.3455761969089508, "reward_total_mean": 0.8526638150215149, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9771455526351929, "rewards/meter/std": 0.02793486975133419, "rewards/total_composite/mean": 0.8526638150215149, "rewards/total_composite/std": 0.3455761969089508, "sampling/importance_sampling_ratio/max": 1.9315763711929321, "sampling/importance_sampling_ratio/mean": 1.0416526794433594, "sampling/importance_sampling_ratio/min": 0.13522395491600037, "sampling/sampling_logp_difference/max": 2.0008230209350586, "sampling/sampling_logp_difference/mean": 0.21847626566886902, "step": 119 }, { "clip_ratio/high_max": 0.0674145333468914, "clip_ratio/high_mean": 0.0674145333468914, "clip_ratio/low_mean": 0.16581876948475838, "clip_ratio/low_min": 0.16581876948475838, "clip_ratio/region_mean": 0.23323330283164978, "completions/clipped_ratio": 0.0, "completions/max_length": 112.0, "completions/max_terminated_length": 112.0, "completions/mean_length": 72.875, "completions/mean_terminated_length": 72.875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 2.9914455711841583, "epoch": 0.004633920296570899, "frac_reward_zero_std": 0.0, "grad_norm": 13.367688179016113, "learning_rate": 9.63939393939394e-06, "loss": 0.1424, "num_tokens": 248367.0, "reward": 0.30540409684181213, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3219662308692932, "reward_meter_std": 0.37545910477638245, "reward_std": 0.3886590600013733, "reward_total_composite_mean": 0.30540409684181213, "reward_total_composite_std": 0.3886590600013733, "reward_total_mean": 0.30540409684181213, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3219662308692932, "rewards/meter/std": 0.37545910477638245, "rewards/total_composite/mean": 0.30540409684181213, "rewards/total_composite/std": 0.3886590600013733, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0693588256835938, "sampling/importance_sampling_ratio/min": 0.1549416035413742, "sampling/sampling_logp_difference/max": 1.8647069931030273, "sampling/sampling_logp_difference/mean": 0.23424479365348816, "step": 120 }, { "clip_ratio/high_max": 0.11291690729558468, "clip_ratio/high_mean": 0.11291690729558468, "clip_ratio/low_mean": 0.07312539592385292, "clip_ratio/low_min": 0.07312539592385292, "clip_ratio/region_mean": 0.1860423032194376, "completions/clipped_ratio": 0.0, "completions/max_length": 174.0, "completions/max_terminated_length": 174.0, "completions/mean_length": 164.5, "completions/mean_terminated_length": 164.5, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "entropy": 1.8652900010347366, "epoch": 0.004672536299042323, "frac_reward_zero_std": 0.0, "grad_norm": 8.033286094665527, "learning_rate": 9.636363636363638e-06, "loss": 0.0533, "num_tokens": 251459.0, "reward": 0.7737554907798767, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7737554907798767, "reward_meter_std": 0.31233546137809753, "reward_std": 0.31233546137809753, "reward_total_composite_mean": 0.7737554907798767, "reward_total_composite_std": 0.31233546137809753, "reward_total_mean": 0.7737554907798767, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7737554907798767, "rewards/meter/std": 0.31233546137809753, "rewards/total_composite/mean": 0.7737554907798767, "rewards/total_composite/std": 0.31233546137809753, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0243695974349976, "sampling/importance_sampling_ratio/min": 0.1459781974554062, "sampling/sampling_logp_difference/max": 1.9242980480194092, "sampling/sampling_logp_difference/mean": 0.17217814922332764, "step": 121 }, { "clip_ratio/high_max": 0.10151049681007862, "clip_ratio/high_mean": 0.10151049681007862, "clip_ratio/low_mean": 0.12684562429785728, "clip_ratio/low_min": 0.12684562429785728, "clip_ratio/region_mean": 0.2283561211079359, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 29.375, "completions/mean_terminated_length": 29.375, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 1.7652578055858612, "epoch": 0.004711152301513748, "frac_reward_zero_std": 0.0, "grad_norm": 21.118961334228516, "learning_rate": 9.633333333333335e-06, "loss": 0.0725, "num_tokens": 252974.0, "reward": 0.4754229187965393, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4754229187965393, "reward_meter_std": 0.4443075954914093, "reward_std": 0.4443075358867645, "reward_total_composite_mean": 0.4754229187965393, "reward_total_composite_std": 0.4443075954914093, "reward_total_mean": 0.4754229187965393, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4754229187965393, "rewards/meter/std": 0.4443075954914093, "rewards/total_composite/mean": 0.4754229187965393, "rewards/total_composite/std": 0.4443075954914093, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0281100273132324, "sampling/importance_sampling_ratio/min": 0.25924697518348694, "sampling/sampling_logp_difference/max": 1.3499740362167358, "sampling/sampling_logp_difference/mean": 0.2008654922246933, "step": 122 }, { "clip_ratio/high_max": 0.14564990997314453, "clip_ratio/high_mean": 0.14564990997314453, "clip_ratio/low_mean": 0.033740474842488766, "clip_ratio/low_min": 0.033740474842488766, "clip_ratio/region_mean": 0.1793903848156333, "completions/clipped_ratio": 0.0, "completions/max_length": 131.0, "completions/max_terminated_length": 131.0, "completions/mean_length": 106.625, "completions/mean_terminated_length": 106.625, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 2.5758557319641113, "epoch": 0.004749768303985172, "frac_reward_zero_std": 0.0, "grad_norm": 10.26867961883545, "learning_rate": 9.63030303030303e-06, "loss": -0.0038, "num_tokens": 255275.0, "reward": 0.9178913831710815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9178913831710815, "reward_meter_std": 0.17877134680747986, "reward_std": 0.17877133190631866, "reward_total_composite_mean": 0.9178913831710815, "reward_total_composite_std": 0.17877134680747986, "reward_total_mean": 0.9178913831710815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9178913831710815, "rewards/meter/std": 0.17877134680747986, "rewards/total_composite/mean": 0.9178913831710815, "rewards/total_composite/std": 0.17877134680747986, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0357023477554321, "sampling/importance_sampling_ratio/min": 0.2035515457391739, "sampling/sampling_logp_difference/max": 1.5918359756469727, "sampling/sampling_logp_difference/mean": 0.20389947295188904, "step": 123 }, { "clip_ratio/high_max": 0.07448212616145611, "clip_ratio/high_mean": 0.07448212616145611, "clip_ratio/low_mean": 0.09205890912562609, "clip_ratio/low_min": 0.09205890912562609, "clip_ratio/region_mean": 0.1665410352870822, "completions/clipped_ratio": 0.0, "completions/max_length": 474.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 340.875, "completions/mean_terminated_length": 340.875, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 3.0468600690364838, "epoch": 0.004788384306456596, "frac_reward_zero_std": 0.0, "grad_norm": 4.5853095054626465, "learning_rate": 9.627272727272728e-06, "loss": 0.0314, "num_tokens": 259818.0, "reward": 0.3518419861793518, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.1679842174053192, "reward_meter_mean": 0.5577775239944458, "reward_meter_std": 0.29362478852272034, "reward_std": 0.3233281970024109, "reward_total_composite_mean": 0.3518419861793518, "reward_total_composite_std": 0.3233281672000885, "reward_total_mean": 0.3518419861793518, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.1679842174053192, "rewards/meter/mean": 0.5577775239944458, "rewards/meter/std": 0.29362478852272034, "rewards/total_composite/mean": 0.3518419861793518, "rewards/total_composite/std": 0.3233281672000885, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.033653974533081, "sampling/importance_sampling_ratio/min": 0.18730992078781128, "sampling/sampling_logp_difference/max": 1.6749906539916992, "sampling/sampling_logp_difference/mean": 0.2014904022216797, "step": 124 }, { "clip_ratio/high_max": 0.15175942331552505, "clip_ratio/high_mean": 0.15175942331552505, "clip_ratio/low_mean": 0.03829259052872658, "clip_ratio/low_min": 0.03829259052872658, "clip_ratio/region_mean": 0.19005201384425163, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 54.75, "completions/mean_terminated_length": 54.75, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 2.44717738032341, "epoch": 0.00482700030892802, "frac_reward_zero_std": 0.0, "grad_norm": 17.506608963012695, "learning_rate": 9.624242424242425e-06, "loss": 0.1625, "num_tokens": 261528.0, "reward": 0.7977160215377808, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7977160215377808, "reward_meter_std": 0.3083810806274414, "reward_std": 0.308381050825119, "reward_total_composite_mean": 0.7977160215377808, "reward_total_composite_std": 0.3083810806274414, "reward_total_mean": 0.7977160215377808, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7977160215377808, "rewards/meter/std": 0.3083810806274414, "rewards/total_composite/mean": 0.7977160215377808, "rewards/total_composite/std": 0.3083810806274414, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0523124933242798, "sampling/importance_sampling_ratio/min": 0.2609032988548279, "sampling/sampling_logp_difference/max": 1.3436055183410645, "sampling/sampling_logp_difference/mean": 0.22570040822029114, "step": 125 }, { "clip_ratio/high_max": 0.026411979109980166, "clip_ratio/high_mean": 0.026411979109980166, "clip_ratio/low_mean": 0.10175232589244843, "clip_ratio/low_min": 0.10175232589244843, "clip_ratio/region_mean": 0.1281643050024286, "completions/clipped_ratio": 0.0, "completions/max_length": 153.0, "completions/max_terminated_length": 153.0, "completions/mean_length": 108.125, "completions/mean_terminated_length": 108.125, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 1.7009220086038113, "epoch": 0.004865616311399444, "frac_reward_zero_std": 0.0, "grad_norm": 13.348600387573242, "learning_rate": 9.621212121212122e-06, "loss": 0.2206, "num_tokens": 263761.0, "reward": 0.5052797198295593, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9285714626312256, "reward_count_adherence_std": 0.07636035233736038, "reward_meter_mean": 0.5293493866920471, "reward_meter_std": 0.396657794713974, "reward_std": 0.39642462134361267, "reward_total_composite_mean": 0.5052797198295593, "reward_total_composite_std": 0.39642462134361267, "reward_total_mean": 0.5052797198295593, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9285714626312256, "rewards/count_adherence/std": 0.07636035233736038, "rewards/meter/mean": 0.5293493866920471, "rewards/meter/std": 0.396657794713974, "rewards/total_composite/mean": 0.5052797198295593, "rewards/total_composite/std": 0.39642462134361267, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.02097487449646, "sampling/importance_sampling_ratio/min": 0.06007884442806244, "sampling/sampling_logp_difference/max": 2.8120975494384766, "sampling/sampling_logp_difference/mean": 0.18913644552230835, "step": 126 }, { "clip_ratio/high_max": 0.08253205195069313, "clip_ratio/high_mean": 0.08253205195069313, "clip_ratio/low_mean": 0.11877263803035021, "clip_ratio/low_min": 0.11877263803035021, "clip_ratio/region_mean": 0.20130468998104334, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 48.0, "completions/mean_terminated_length": 48.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 2.404453784227371, "epoch": 0.004904232313870868, "frac_reward_zero_std": 0.0, "grad_norm": 14.61343002319336, "learning_rate": 9.61818181818182e-06, "loss": 0.0506, "num_tokens": 265393.0, "reward": 0.3801646828651428, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3801646828651428, "reward_meter_std": 0.4593394994735718, "reward_std": 0.4593394994735718, "reward_total_composite_mean": 0.3801646828651428, "reward_total_composite_std": 0.4593394994735718, "reward_total_mean": 0.3801646828651428, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3801646828651428, "rewards/meter/std": 0.4593394994735718, "rewards/total_composite/mean": 0.3801646828651428, "rewards/total_composite/std": 0.4593394994735718, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0501198768615723, "sampling/importance_sampling_ratio/min": 0.3086847960948944, "sampling/sampling_logp_difference/max": 1.1754345893859863, "sampling/sampling_logp_difference/mean": 0.19850540161132812, "step": 127 }, { "clip_ratio/high_max": 0.14333923161029816, "clip_ratio/high_mean": 0.14333923161029816, "clip_ratio/low_mean": 0.04620295576751232, "clip_ratio/low_min": 0.04620295576751232, "clip_ratio/region_mean": 0.18954218737781048, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 55.125, "completions/mean_terminated_length": 55.125, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 2.133269816637039, "epoch": 0.004942848316342292, "frac_reward_zero_std": 0.0, "grad_norm": 14.53402042388916, "learning_rate": 9.615151515151517e-06, "loss": 0.035, "num_tokens": 267018.0, "reward": 0.7447742223739624, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7447742223739624, "reward_meter_std": 0.3583148717880249, "reward_std": 0.3583148717880249, "reward_total_composite_mean": 0.7447742223739624, "reward_total_composite_std": 0.3583148717880249, "reward_total_mean": 0.7447742223739624, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7447742223739624, "rewards/meter/std": 0.3583148717880249, "rewards/total_composite/mean": 0.7447742223739624, "rewards/total_composite/std": 0.3583148717880249, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.042881965637207, "sampling/importance_sampling_ratio/min": 0.2833141088485718, "sampling/sampling_logp_difference/max": 1.2611989974975586, "sampling/sampling_logp_difference/mean": 0.20606549084186554, "step": 128 }, { "clip_ratio/high_max": 0.10729072242975235, "clip_ratio/high_mean": 0.10729072242975235, "clip_ratio/low_mean": 0.05078725144267082, "clip_ratio/low_min": 0.05078725144267082, "clip_ratio/region_mean": 0.15807797387242317, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 24.0, "completions/mean_terminated_length": 24.0, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 1.7079626321792603, "epoch": 0.004981464318813716, "frac_reward_zero_std": 0.0, "grad_norm": 23.683666229248047, "learning_rate": 9.612121212121212e-06, "loss": 0.2607, "num_tokens": 268338.0, "reward": 0.6895738840103149, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6895738840103149, "reward_meter_std": 0.3600609004497528, "reward_std": 0.3600608706474304, "reward_total_composite_mean": 0.6895738840103149, "reward_total_composite_std": 0.3600609004497528, "reward_total_mean": 0.6895738840103149, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6895738840103149, "rewards/meter/std": 0.3600609004497528, "rewards/total_composite/mean": 0.6895738840103149, "rewards/total_composite/std": 0.3600609004497528, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.022581696510315, "sampling/importance_sampling_ratio/min": 0.2624211013317108, "sampling/sampling_logp_difference/max": 1.3378047943115234, "sampling/sampling_logp_difference/mean": 0.21219754219055176, "step": 129 }, { "clip_ratio/high_max": 0.07609842158854008, "clip_ratio/high_mean": 0.07609842158854008, "clip_ratio/low_mean": 0.10636867955327034, "clip_ratio/low_min": 0.10636867955327034, "clip_ratio/region_mean": 0.18246710114181042, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 53.0, "completions/mean_terminated_length": 53.0, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 2.128389284014702, "epoch": 0.0050200803212851405, "frac_reward_zero_std": 0.0, "grad_norm": 13.415695190429688, "learning_rate": 9.60909090909091e-06, "loss": 0.0564, "num_tokens": 270058.0, "reward": 0.2599804103374481, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2599804103374481, "reward_meter_std": 0.33920830488204956, "reward_std": 0.33920830488204956, "reward_total_composite_mean": 0.2599804103374481, "reward_total_composite_std": 0.33920830488204956, "reward_total_mean": 0.2599804103374481, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2599804103374481, "rewards/meter/std": 0.33920830488204956, "rewards/total_composite/mean": 0.2599804103374481, "rewards/total_composite/std": 0.33920830488204956, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0430470705032349, "sampling/importance_sampling_ratio/min": 0.2772029638290405, "sampling/sampling_logp_difference/max": 1.2830052375793457, "sampling/sampling_logp_difference/mean": 0.2094995379447937, "step": 130 }, { "clip_ratio/high_max": 0.1310401326045394, "clip_ratio/high_mean": 0.1310401326045394, "clip_ratio/low_mean": 0.07387228682637215, "clip_ratio/low_min": 0.07387228682637215, "clip_ratio/region_mean": 0.20491241943091154, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 54.75, "completions/mean_terminated_length": 54.75, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 2.6511287838220596, "epoch": 0.005058696323756565, "frac_reward_zero_std": 0.0, "grad_norm": 17.473791122436523, "learning_rate": 9.606060606060607e-06, "loss": 0.0505, "num_tokens": 271776.0, "reward": 0.6479246616363525, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6479246616363525, "reward_meter_std": 0.4171454906463623, "reward_std": 0.41714543104171753, "reward_total_composite_mean": 0.6479246616363525, "reward_total_composite_std": 0.4171454906463623, "reward_total_mean": 0.6479246616363525, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6479246616363525, "rewards/meter/std": 0.4171454906463623, "rewards/total_composite/mean": 0.6479246616363525, "rewards/total_composite/std": 0.4171454906463623, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0370509624481201, "sampling/importance_sampling_ratio/min": 0.2412000447511673, "sampling/sampling_logp_difference/max": 1.422128677368164, "sampling/sampling_logp_difference/mean": 0.21375833451747894, "step": 131 }, { "clip_ratio/high_max": 0.07182539906352758, "clip_ratio/high_mean": 0.07182539906352758, "clip_ratio/low_mean": 0.09321646392345428, "clip_ratio/low_min": 0.09321646392345428, "clip_ratio/region_mean": 0.16504186298698187, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 45.25, "completions/mean_terminated_length": 45.25, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 2.1410828977823257, "epoch": 0.005097312326227989, "frac_reward_zero_std": 0.0, "grad_norm": 15.81747817993164, "learning_rate": 9.603030303030304e-06, "loss": 0.0909, "num_tokens": 273490.0, "reward": 0.2023978978395462, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2023978978395462, "reward_meter_std": 0.27117809653282166, "reward_std": 0.27117806673049927, "reward_total_composite_mean": 0.2023978978395462, "reward_total_composite_std": 0.27117809653282166, "reward_total_mean": 0.2023978978395462, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2023978978395462, "rewards/meter/std": 0.27117809653282166, "rewards/total_composite/mean": 0.2023978978395462, "rewards/total_composite/std": 0.27117809653282166, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0494537353515625, "sampling/importance_sampling_ratio/min": 0.2672046720981598, "sampling/sampling_logp_difference/max": 1.3197402954101562, "sampling/sampling_logp_difference/mean": 0.2216556966304779, "step": 132 }, { "clip_ratio/high_max": 0.10300109349191189, "clip_ratio/high_mean": 0.10300109349191189, "clip_ratio/low_mean": 0.10132601391524076, "clip_ratio/low_min": 0.10132601391524076, "clip_ratio/region_mean": 0.20432710740715265, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 49.0, "completions/mean_terminated_length": 49.0, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 2.9459713995456696, "epoch": 0.005135928328699413, "frac_reward_zero_std": 0.0, "grad_norm": 15.763381004333496, "learning_rate": 9.600000000000001e-06, "loss": -0.0793, "num_tokens": 275138.0, "reward": 0.5731572508811951, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5731572508811951, "reward_meter_std": 0.42340850830078125, "reward_std": 0.42340850830078125, "reward_total_composite_mean": 0.5731572508811951, "reward_total_composite_std": 0.42340850830078125, "reward_total_mean": 0.5731572508811951, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5731572508811951, "rewards/meter/std": 0.42340850830078125, "rewards/total_composite/mean": 0.5731572508811951, "rewards/total_composite/std": 0.42340850830078125, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0395585298538208, "sampling/importance_sampling_ratio/min": 0.254414439201355, "sampling/sampling_logp_difference/max": 1.368790626525879, "sampling/sampling_logp_difference/mean": 0.21350626647472382, "step": 133 }, { "clip_ratio/high_max": 0.1634557507932186, "clip_ratio/high_mean": 0.1634557507932186, "clip_ratio/low_mean": 0.05536198429763317, "clip_ratio/low_min": 0.05536198429763317, "clip_ratio/region_mean": 0.21881773509085178, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 57.125, "completions/mean_terminated_length": 57.125, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 2.2369899600744247, "epoch": 0.005174544331170837, "frac_reward_zero_std": 0.0, "grad_norm": 15.412650108337402, "learning_rate": 9.596969696969699e-06, "loss": -0.0251, "num_tokens": 276875.0, "reward": 0.7134405970573425, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7134405970573425, "reward_meter_std": 0.36172330379486084, "reward_std": 0.36172330379486084, "reward_total_composite_mean": 0.7134405970573425, "reward_total_composite_std": 0.36172330379486084, "reward_total_mean": 0.7134405970573425, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7134405970573425, "rewards/meter/std": 0.36172330379486084, "rewards/total_composite/mean": 0.7134405970573425, "rewards/total_composite/std": 0.36172330379486084, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0319643020629883, "sampling/importance_sampling_ratio/min": 0.18233875930309296, "sampling/sampling_logp_difference/max": 1.7018890380859375, "sampling/sampling_logp_difference/mean": 0.20017951726913452, "step": 134 }, { "clip_ratio/high_max": 0.07667344622313976, "clip_ratio/high_mean": 0.07667344622313976, "clip_ratio/low_mean": 0.09396191779524088, "clip_ratio/low_min": 0.09396191779524088, "clip_ratio/region_mean": 0.17063536401838064, "completions/clipped_ratio": 0.0, "completions/max_length": 281.0, "completions/max_terminated_length": 281.0, "completions/mean_length": 194.875, "completions/mean_terminated_length": 194.875, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 3.0369058549404144, "epoch": 0.005213160333642261, "frac_reward_zero_std": 0.0, "grad_norm": 6.406003952026367, "learning_rate": 9.593939393939394e-06, "loss": 0.009, "num_tokens": 279842.0, "reward": 0.3583834171295166, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9464285969734192, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.3760119676589966, "reward_meter_std": 0.4099160134792328, "reward_std": 0.4076659679412842, "reward_total_composite_mean": 0.3583834171295166, "reward_total_composite_std": 0.4076659679412842, "reward_total_mean": 0.3583834171295166, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9464285969734192, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.3760119676589966, "rewards/meter/std": 0.4099160134792328, "rewards/total_composite/mean": 0.3583834171295166, "rewards/total_composite/std": 0.4076659679412842, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0401254892349243, "sampling/importance_sampling_ratio/min": 0.21446330845355988, "sampling/sampling_logp_difference/max": 1.539616584777832, "sampling/sampling_logp_difference/mean": 0.1984342634677887, "step": 135 }, { "clip_ratio/high_max": 0.11762434057891369, "clip_ratio/high_mean": 0.11762434057891369, "clip_ratio/low_mean": 0.06984265986829996, "clip_ratio/low_min": 0.06984265986829996, "clip_ratio/region_mean": 0.18746700044721365, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 51.25, "completions/mean_terminated_length": 51.25, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 2.2263315618038177, "epoch": 0.005251776336113685, "frac_reward_zero_std": 0.0, "grad_norm": 13.780417442321777, "learning_rate": 9.590909090909091e-06, "loss": -0.0186, "num_tokens": 281484.0, "reward": 0.6177759170532227, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6177759170532227, "reward_meter_std": 0.46374326944351196, "reward_std": 0.4637432098388672, "reward_total_composite_mean": 0.6177759170532227, "reward_total_composite_std": 0.46374326944351196, "reward_total_mean": 0.6177759170532227, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6177759170532227, "rewards/meter/std": 0.46374326944351196, "rewards/total_composite/mean": 0.6177759170532227, "rewards/total_composite/std": 0.46374326944351196, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0363706350326538, "sampling/importance_sampling_ratio/min": 0.15855665504932404, "sampling/sampling_logp_difference/max": 1.8416433334350586, "sampling/sampling_logp_difference/mean": 0.210972860455513, "step": 136 }, { "clip_ratio/high_max": 0.1296451175585389, "clip_ratio/high_mean": 0.1296451175585389, "clip_ratio/low_mean": 0.037981835193932056, "clip_ratio/low_min": 0.037981835193932056, "clip_ratio/region_mean": 0.16762695275247097, "completions/clipped_ratio": 0.0, "completions/max_length": 45.0, "completions/max_terminated_length": 45.0, "completions/mean_length": 32.25, "completions/mean_terminated_length": 32.25, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 1.1308462470769882, "epoch": 0.005290392338585109, "frac_reward_zero_std": 0.0, "grad_norm": 21.104068756103516, "learning_rate": 9.587878787878789e-06, "loss": 0.0346, "num_tokens": 282990.0, "reward": 0.6098703145980835, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6098703145980835, "reward_meter_std": 0.3275175094604492, "reward_std": 0.3275175392627716, "reward_total_composite_mean": 0.6098703145980835, "reward_total_composite_std": 0.3275175094604492, "reward_total_mean": 0.6098703145980835, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6098703145980835, "rewards/meter/std": 0.3275175094604492, "rewards/total_composite/mean": 0.6098703145980835, "rewards/total_composite/std": 0.3275175094604492, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9829218983650208, "sampling/importance_sampling_ratio/min": 0.2552906572818756, "sampling/sampling_logp_difference/max": 1.3653526306152344, "sampling/sampling_logp_difference/mean": 0.15988147258758545, "step": 137 }, { "clip_ratio/high_max": 0.08452141657471657, "clip_ratio/high_mean": 0.08452141657471657, "clip_ratio/low_mean": 0.11034664139151573, "clip_ratio/low_min": 0.11034664139151573, "clip_ratio/region_mean": 0.1948680579662323, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 27.125, "completions/mean_terminated_length": 27.125, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 2.293392300605774, "epoch": 0.005329008341056534, "frac_reward_zero_std": 0.0, "grad_norm": 19.138347625732422, "learning_rate": 9.584848484848486e-06, "loss": 0.0749, "num_tokens": 284471.0, "reward": 0.3993714153766632, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.3993714153766632, "reward_meter_std": 0.3401057720184326, "reward_std": 0.34010574221611023, "reward_total_composite_mean": 0.3993714153766632, "reward_total_composite_std": 0.3401057720184326, "reward_total_mean": 0.3993714153766632, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.3993714153766632, "rewards/meter/std": 0.3401057720184326, "rewards/total_composite/mean": 0.3993714153766632, "rewards/total_composite/std": 0.3401057720184326, "sampling/importance_sampling_ratio/max": 1.970318078994751, "sampling/importance_sampling_ratio/mean": 1.050313949584961, "sampling/importance_sampling_ratio/min": 0.4154578745365143, "sampling/sampling_logp_difference/max": 0.8783740997314453, "sampling/sampling_logp_difference/mean": 0.19733576476573944, "step": 138 }, { "clip_ratio/high_max": 0.13470745086669922, "clip_ratio/high_mean": 0.13470745086669922, "clip_ratio/low_mean": 0.03914893604815006, "clip_ratio/low_min": 0.03914893604815006, "clip_ratio/region_mean": 0.17385638691484928, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 292.0, "completions/mean_terminated_length": 292.0, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 3.159882366657257, "epoch": 0.005367624343527958, "frac_reward_zero_std": 0.0, "grad_norm": 4.607113838195801, "learning_rate": 9.581818181818181e-06, "loss": -0.0421, "num_tokens": 288591.0, "reward": 0.7447197437286377, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_meter_mean": 0.8927704095840454, "reward_meter_std": 0.13047616183757782, "reward_std": 0.3232608437538147, "reward_total_composite_mean": 0.7447197437286377, "reward_total_composite_std": 0.3232608437538147, "reward_total_mean": 0.7447197437286377, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/meter/mean": 0.8927704095840454, "rewards/meter/std": 0.13047616183757782, "rewards/total_composite/mean": 0.7447197437286377, "rewards/total_composite/std": 0.3232608437538147, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0477176904678345, "sampling/importance_sampling_ratio/min": 0.2002381682395935, "sampling/sampling_logp_difference/max": 1.6082477569580078, "sampling/sampling_logp_difference/mean": 0.19803562760353088, "step": 139 }, { "clip_ratio/high_max": 0.10052786208689213, "clip_ratio/high_mean": 0.10052786208689213, "clip_ratio/low_mean": 0.084076764062047, "clip_ratio/low_min": 0.084076764062047, "clip_ratio/region_mean": 0.18460462614893913, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 304.5, "completions/mean_terminated_length": 304.5, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "entropy": 2.8953827619552612, "epoch": 0.0054062403459993824, "frac_reward_zero_std": 0.0, "grad_norm": 5.223753929138184, "learning_rate": 9.57878787878788e-06, "loss": 0.0406, "num_tokens": 292571.0, "reward": 0.5782480835914612, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.0534522607922554, "reward_meter_mean": 0.609697163105011, "reward_meter_std": 0.17199508845806122, "reward_std": 0.16025982797145844, "reward_total_composite_mean": 0.5782480835914612, "reward_total_composite_std": 0.16025982797145844, "reward_total_mean": 0.5782480835914612, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.0534522607922554, "rewards/meter/mean": 0.609697163105011, "rewards/meter/std": 0.17199508845806122, "rewards/total_composite/mean": 0.5782480835914612, "rewards/total_composite/std": 0.16025982797145844, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.034401535987854, "sampling/importance_sampling_ratio/min": 0.14130491018295288, "sampling/sampling_logp_difference/max": 1.9568352699279785, "sampling/sampling_logp_difference/mean": 0.19511224329471588, "step": 140 }, { "clip_ratio/high_max": 0.06570382788777351, "clip_ratio/high_mean": 0.06570382788777351, "clip_ratio/low_mean": 0.08571217767894268, "clip_ratio/low_min": 0.08571217767894268, "clip_ratio/region_mean": 0.1514160055667162, "completions/clipped_ratio": 0.0, "completions/max_length": 97.0, "completions/max_terminated_length": 97.0, "completions/mean_length": 86.5, "completions/mean_terminated_length": 86.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 2.2329191118478775, "epoch": 0.0054448563484708066, "frac_reward_zero_std": 0.0, "grad_norm": 10.004317283630371, "learning_rate": 9.575757575757576e-06, "loss": -0.0092, "num_tokens": 294679.0, "reward": 0.414273202419281, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.414273202419281, "reward_meter_std": 0.41314172744750977, "reward_std": 0.41314172744750977, "reward_total_composite_mean": 0.414273202419281, "reward_total_composite_std": 0.41314172744750977, "reward_total_mean": 0.414273202419281, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.414273202419281, "rewards/meter/std": 0.41314172744750977, "rewards/total_composite/mean": 0.414273202419281, "rewards/total_composite/std": 0.41314172744750977, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0314459800720215, "sampling/importance_sampling_ratio/min": 0.35211867094039917, "sampling/sampling_logp_difference/max": 1.0437870025634766, "sampling/sampling_logp_difference/mean": 0.17779497802257538, "step": 141 }, { "clip_ratio/high_max": 0.06938760355114937, "clip_ratio/high_mean": 0.06938760355114937, "clip_ratio/low_mean": 0.11229986138641834, "clip_ratio/low_min": 0.11229986138641834, "clip_ratio/region_mean": 0.1816874649375677, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 47.625, "completions/mean_terminated_length": 47.625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 2.629018932580948, "epoch": 0.005483472350942231, "frac_reward_zero_std": 0.0, "grad_norm": 13.359566688537598, "learning_rate": 9.572727272727273e-06, "loss": -0.0499, "num_tokens": 296372.0, "reward": 0.4189656376838684, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4189656376838684, "reward_meter_std": 0.3781832754611969, "reward_std": 0.3781832456588745, "reward_total_composite_mean": 0.4189656376838684, "reward_total_composite_std": 0.3781832754611969, "reward_total_mean": 0.4189656376838684, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4189656376838684, "rewards/meter/std": 0.3781832754611969, "rewards/total_composite/mean": 0.4189656376838684, "rewards/total_composite/std": 0.3781832754611969, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0515532493591309, "sampling/importance_sampling_ratio/min": 0.1484287977218628, "sampling/sampling_logp_difference/max": 1.9076499938964844, "sampling/sampling_logp_difference/mean": 0.22827744483947754, "step": 142 }, { "clip_ratio/high_max": 0.0911912564188242, "clip_ratio/high_mean": 0.0911912564188242, "clip_ratio/low_mean": 0.096000537276268, "clip_ratio/low_min": 0.096000537276268, "clip_ratio/region_mean": 0.1871917936950922, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 84.875, "completions/mean_terminated_length": 84.875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 2.4079796075820923, "epoch": 0.005522088353413655, "frac_reward_zero_std": 0.0, "grad_norm": 9.326652526855469, "learning_rate": 9.56969696969697e-06, "loss": -0.005, "num_tokens": 298243.0, "reward": 0.5255526304244995, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5255526304244995, "reward_meter_std": 0.3781281113624573, "reward_std": 0.3781280815601349, "reward_total_composite_mean": 0.5255526304244995, "reward_total_composite_std": 0.3781281113624573, "reward_total_mean": 0.5255526304244995, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5255526304244995, "rewards/meter/std": 0.3781281113624573, "rewards/total_composite/mean": 0.5255526304244995, "rewards/total_composite/std": 0.3781281113624573, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0423548221588135, "sampling/importance_sampling_ratio/min": 0.2729393243789673, "sampling/sampling_logp_difference/max": 1.2985057830810547, "sampling/sampling_logp_difference/mean": 0.20440031588077545, "step": 143 }, { "clip_ratio/high_max": 0.08965095691382885, "clip_ratio/high_mean": 0.08965095691382885, "clip_ratio/low_mean": 0.09224239364266396, "clip_ratio/low_min": 0.09224239364266396, "clip_ratio/region_mean": 0.1818933505564928, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 2.5128345042467117, "epoch": 0.005560704355885079, "frac_reward_zero_std": 0.0, "grad_norm": 12.313883781433105, "learning_rate": 9.566666666666668e-06, "loss": 0.1133, "num_tokens": 299963.0, "reward": 0.6259543299674988, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6259543299674988, "reward_meter_std": 0.40781956911087036, "reward_std": 0.40781959891319275, "reward_total_composite_mean": 0.6259543299674988, "reward_total_composite_std": 0.40781956911087036, "reward_total_mean": 0.6259543299674988, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6259543299674988, "rewards/meter/std": 0.40781956911087036, "rewards/total_composite/mean": 0.6259543299674988, "rewards/total_composite/std": 0.40781956911087036, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0449057817459106, "sampling/importance_sampling_ratio/min": 0.2863829731941223, "sampling/sampling_logp_difference/max": 1.2504253387451172, "sampling/sampling_logp_difference/mean": 0.1947186291217804, "step": 144 }, { "clip_ratio/high_max": 0.12736221216619015, "clip_ratio/high_mean": 0.12736221216619015, "clip_ratio/low_mean": 0.07416711933910847, "clip_ratio/low_min": 0.07416711933910847, "clip_ratio/region_mean": 0.20152933150529861, "completions/clipped_ratio": 0.0, "completions/max_length": 130.0, "completions/max_terminated_length": 130.0, "completions/mean_length": 108.0, "completions/mean_terminated_length": 108.0, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 2.594912603497505, "epoch": 0.005599320358356503, "frac_reward_zero_std": 0.0, "grad_norm": 9.220316886901855, "learning_rate": 9.563636363636365e-06, "loss": -0.0099, "num_tokens": 302259.0, "reward": 0.6794947385787964, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_meter_mean": 0.7156338095664978, "reward_meter_std": 0.3621709942817688, "reward_std": 0.3623289465904236, "reward_total_composite_mean": 0.6794947385787964, "reward_total_composite_std": 0.3623289465904236, "reward_total_mean": 0.6794947385787964, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/meter/mean": 0.7156338095664978, "rewards/meter/std": 0.3621709942817688, "rewards/total_composite/mean": 0.6794947385787964, "rewards/total_composite/std": 0.3623289465904236, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0389753580093384, "sampling/importance_sampling_ratio/min": 0.19741055369377136, "sampling/sampling_logp_difference/max": 1.6224696636199951, "sampling/sampling_logp_difference/mean": 0.20363517105579376, "step": 145 }, { "clip_ratio/high_max": 0.06477605737745762, "clip_ratio/high_mean": 0.06477605737745762, "clip_ratio/low_mean": 0.09793206304311752, "clip_ratio/low_min": 0.09793206304311752, "clip_ratio/region_mean": 0.16270812042057514, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 85.0, "completions/mean_terminated_length": 85.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.9736258126795292, "epoch": 0.005637936360827927, "frac_reward_zero_std": 0.0, "grad_norm": 21.128833770751953, "learning_rate": 9.56060606060606e-06, "loss": 0.0802, "num_tokens": 304531.0, "reward": 0.4409075975418091, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.5389710068702698, "reward_meter_std": 0.36102810502052307, "reward_std": 0.28550228476524353, "reward_total_composite_mean": 0.4409075975418091, "reward_total_composite_std": 0.28550228476524353, "reward_total_mean": 0.4409075975418091, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.5389710068702698, "rewards/meter/std": 0.36102810502052307, "rewards/total_composite/mean": 0.4409075975418091, "rewards/total_composite/std": 0.28550228476524353, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0029895305633545, "sampling/importance_sampling_ratio/min": 0.029469041153788567, "sampling/sampling_logp_difference/max": 3.5244150161743164, "sampling/sampling_logp_difference/mean": 0.20141077041625977, "step": 146 }, { "clip_ratio/high_max": 0.11702838353812695, "clip_ratio/high_mean": 0.11702838353812695, "clip_ratio/low_mean": 0.07888335548341274, "clip_ratio/low_min": 0.07888335548341274, "clip_ratio/region_mean": 0.1959117390215397, "completions/clipped_ratio": 0.0, "completions/max_length": 112.0, "completions/max_terminated_length": 112.0, "completions/mean_length": 93.125, "completions/mean_terminated_length": 93.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 2.3368784189224243, "epoch": 0.005676552363299351, "frac_reward_zero_std": 0.0, "grad_norm": 9.888257026672363, "learning_rate": 9.55757575757576e-06, "loss": 0.0747, "num_tokens": 306668.0, "reward": 0.6462192535400391, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7655929327011108, "reward_meter_std": 0.3509427309036255, "reward_std": 0.43067827820777893, "reward_total_composite_mean": 0.6462192535400391, "reward_total_composite_std": 0.43067827820777893, "reward_total_mean": 0.6462192535400391, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7655929327011108, "rewards/meter/std": 0.3509427309036255, "rewards/total_composite/mean": 0.6462192535400391, "rewards/total_composite/std": 0.43067827820777893, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0265202522277832, "sampling/importance_sampling_ratio/min": 0.2604142129421234, "sampling/sampling_logp_difference/max": 1.3454818725585938, "sampling/sampling_logp_difference/mean": 0.19887900352478027, "step": 147 }, { "clip_ratio/high_max": 0.18514032009989023, "clip_ratio/high_mean": 0.18514032009989023, "clip_ratio/low_mean": 0.02163461595773697, "clip_ratio/low_min": 0.02163461595773697, "clip_ratio/region_mean": 0.2067749360576272, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 53.625, "completions/mean_terminated_length": 53.625, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 2.712312787771225, "epoch": 0.005715168365770775, "frac_reward_zero_std": 0.0, "grad_norm": 13.740280151367188, "learning_rate": 9.554545454545455e-06, "loss": 0.0131, "num_tokens": 308289.0, "reward": 0.8362427353858948, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9531986117362976, "reward_meter_std": 0.048368558287620544, "reward_std": 0.3412637710571289, "reward_total_composite_mean": 0.8362427353858948, "reward_total_composite_std": 0.3412637710571289, "reward_total_mean": 0.8362427353858948, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9531986117362976, "rewards/meter/std": 0.048368558287620544, "rewards/total_composite/mean": 0.8362427353858948, "rewards/total_composite/std": 0.3412637710571289, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0660349130630493, "sampling/importance_sampling_ratio/min": 0.07845636457204819, "sampling/sampling_logp_difference/max": 2.545212745666504, "sampling/sampling_logp_difference/mean": 0.22549496591091156, "step": 148 }, { "clip_ratio/high_max": 0.06303809583187103, "clip_ratio/high_mean": 0.06303809583187103, "clip_ratio/low_mean": 0.052138859406113625, "clip_ratio/low_min": 0.052138859406113625, "clip_ratio/region_mean": 0.11517695523798466, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 72.5, "completions/mean_terminated_length": 72.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.622876338660717, "epoch": 0.005753784368242199, "frac_reward_zero_std": 0.0, "grad_norm": 30.763076782226562, "learning_rate": 9.551515151515152e-06, "loss": 0.1423, "num_tokens": 310181.0, "reward": 0.764354407787323, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.764354407787323, "reward_meter_std": 0.35320577025413513, "reward_std": 0.35320577025413513, "reward_total_composite_mean": 0.764354407787323, "reward_total_composite_std": 0.35320577025413513, "reward_total_mean": 0.764354407787323, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.764354407787323, "rewards/meter/std": 0.35320577025413513, "rewards/total_composite/mean": 0.764354407787323, "rewards/total_composite/std": 0.35320577025413513, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017551183700562, "sampling/importance_sampling_ratio/min": 0.09305361658334732, "sampling/sampling_logp_difference/max": 2.374579429626465, "sampling/sampling_logp_difference/mean": 0.14173762500286102, "step": 149 }, { "clip_ratio/high_max": 0.06511373445391655, "clip_ratio/high_mean": 0.06511373445391655, "clip_ratio/low_mean": 0.11349504999816418, "clip_ratio/low_min": 0.11349504999816418, "clip_ratio/region_mean": 0.17860878445208073, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 255.25, "completions/mean_terminated_length": 255.25, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 3.1247769594192505, "epoch": 0.0057924003707136235, "frac_reward_zero_std": 0.0, "grad_norm": 5.072005748748779, "learning_rate": 9.54848484848485e-06, "loss": -0.0588, "num_tokens": 313791.0, "reward": 0.32547223567962646, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.9861111044883728, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.4467889070510864, "reward_meter_std": 0.29338762164115906, "reward_std": 0.2992675006389618, "reward_total_composite_mean": 0.32547223567962646, "reward_total_composite_std": 0.2992675006389618, "reward_total_mean": 0.32547223567962646, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.9861111044883728, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.4467889070510864, "rewards/meter/std": 0.29338762164115906, "rewards/total_composite/mean": 0.32547223567962646, "rewards/total_composite/std": 0.2992675006389618, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0436980724334717, "sampling/importance_sampling_ratio/min": 0.20569318532943726, "sampling/sampling_logp_difference/max": 1.5813696384429932, "sampling/sampling_logp_difference/mean": 0.20325259864330292, "step": 150 }, { "epoch": 0.0057924003707136235, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.038461538461538464, "eval_completions/max_length": 435.3076923076923, "eval_completions/max_terminated_length": 406.38461538461536, "eval_completions/mean_length": 195.05769230769232, "eval_completions/mean_terminated_length": 182.7815962571364, "eval_completions/min_length": 42.92307692307692, "eval_completions/min_terminated_length": 42.92307692307692, "eval_entropy": 2.4764969715705285, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 313791.0, "eval_reward": 0.4012144311116292, "eval_reward_arabic_clean_mean": 0.9423076923076923, "eval_reward_arabic_clean_std": 0.1631784851734455, "eval_reward_count_adherence_mean": 0.9416975012192359, "eval_reward_count_adherence_std": 0.08222452465158242, "eval_reward_meter_mean": 0.45025052244846636, "eval_reward_meter_std": 0.34984474113354314, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.4012144311116292, "eval_reward_total_composite_std": 0.3435493845206041, "eval_reward_total_mean": 0.4012144311116292, "eval_rewards/arabic_clean/mean": 0.9423076923076923, "eval_rewards/arabic_clean/std": 0.1631784851734455, "eval_rewards/count_adherence/mean": 0.9416975012192359, "eval_rewards/count_adherence/std": 0.08222452465158242, "eval_rewards/meter/mean": 0.45025052244846636, "eval_rewards/meter/std": 0.34984474113354314, "eval_rewards/total_composite/mean": 0.4012144311116292, "eval_rewards/total_composite/std": 0.3435493845206041, "eval_runtime": 80.4379, "eval_samples_per_second": 1.293, "eval_sampling/importance_sampling_ratio/max": 1.593602758187514, "eval_sampling/importance_sampling_ratio/mean": 1.0403164625167847, "eval_sampling/importance_sampling_ratio/min": 0.27021744961921984, "eval_sampling/sampling_logp_difference/max": 1.3218139501718373, "eval_sampling/sampling_logp_difference/mean": 0.14192955310528094, "eval_steps_per_second": 0.162, "step": 150 }, { "clip_ratio/high_max": 0.10548157431185246, "clip_ratio/high_mean": 0.10548157431185246, "clip_ratio/low_mean": 0.05626178905367851, "clip_ratio/low_min": 0.05626178905367851, "clip_ratio/region_mean": 0.16174336336553097, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 328.5, "completions/mean_terminated_length": 328.5, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 2.69022598862648, "epoch": 0.005831016373185048, "frac_reward_zero_std": 0.0, "grad_norm": 4.5317206382751465, "learning_rate": 9.545454545454547e-06, "loss": -0.0419, "num_tokens": 318003.0, "reward": 0.9565757513046265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_meter_mean": 0.9837745428085327, "reward_meter_std": 0.0232550036162138, "reward_std": 0.0579642690718174, "reward_total_composite_mean": 0.9565757513046265, "reward_total_composite_std": 0.05796428769826889, "reward_total_mean": 0.9565757513046265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/meter/mean": 0.9837745428085327, "rewards/meter/std": 0.0232550036162138, "rewards/total_composite/mean": 0.9565757513046265, "rewards/total_composite/std": 0.05796428769826889, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0254464149475098, "sampling/importance_sampling_ratio/min": 0.2590313255786896, "sampling/sampling_logp_difference/max": 1.3508062362670898, "sampling/sampling_logp_difference/mean": 0.18240459263324738, "step": 151 }, { "clip_ratio/high_max": 0.10017775557935238, "clip_ratio/high_mean": 0.10017775557935238, "clip_ratio/low_mean": 0.08865037560462952, "clip_ratio/low_min": 0.08865037560462952, "clip_ratio/region_mean": 0.1888281311839819, "completions/clipped_ratio": 0.0, "completions/max_length": 260.0, "completions/max_terminated_length": 260.0, "completions/mean_length": 222.5, "completions/mean_terminated_length": 222.5, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "entropy": 2.6781843453645706, "epoch": 0.005869632375656472, "frac_reward_zero_std": 0.0, "grad_norm": 6.531708240509033, "learning_rate": 9.542424242424242e-06, "loss": 0.0225, "num_tokens": 321511.0, "reward": 0.6029950380325317, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.6115673780441284, "reward_meter_std": 0.26747360825538635, "reward_std": 0.2708563804626465, "reward_total_composite_mean": 0.6029950380325317, "reward_total_composite_std": 0.2708563804626465, "reward_total_mean": 0.6029950380325317, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.6115673780441284, "rewards/meter/std": 0.26747360825538635, "rewards/total_composite/mean": 0.6029950380325317, "rewards/total_composite/std": 0.2708563804626465, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0410678386688232, "sampling/importance_sampling_ratio/min": 0.20417055487632751, "sampling/sampling_logp_difference/max": 1.5887994766235352, "sampling/sampling_logp_difference/mean": 0.20755378901958466, "step": 152 }, { "clip_ratio/high_max": 0.15194394160062075, "clip_ratio/high_mean": 0.15194394160062075, "clip_ratio/low_mean": 0.05802265927195549, "clip_ratio/low_min": 0.05802265927195549, "clip_ratio/region_mean": 0.20996660087257624, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 50.75, "completions/mean_terminated_length": 50.75, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 2.6396835446357727, "epoch": 0.005908248378127896, "frac_reward_zero_std": 0.0, "grad_norm": 15.966812133789062, "learning_rate": 9.539393939393941e-06, "loss": 0.1928, "num_tokens": 323205.0, "reward": 0.7692294716835022, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7692294716835022, "reward_meter_std": 0.3993333876132965, "reward_std": 0.3993334174156189, "reward_total_composite_mean": 0.7692294716835022, "reward_total_composite_std": 0.3993333876132965, "reward_total_mean": 0.7692294716835022, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7692294716835022, "rewards/meter/std": 0.3993333876132965, "rewards/total_composite/mean": 0.7692294716835022, "rewards/total_composite/std": 0.3993333876132965, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0662130117416382, "sampling/importance_sampling_ratio/min": 0.23212644457817078, "sampling/sampling_logp_difference/max": 1.4604730606079102, "sampling/sampling_logp_difference/mean": 0.21327464282512665, "step": 153 }, { "clip_ratio/high_max": 0.14308988489210606, "clip_ratio/high_mean": 0.14308988489210606, "clip_ratio/low_mean": 0.047645075246691704, "clip_ratio/low_min": 0.047645075246691704, "clip_ratio/region_mean": 0.19073496013879776, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 72.375, "completions/mean_terminated_length": 72.375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 2.7783928215503693, "epoch": 0.005946864380599321, "frac_reward_zero_std": 0.0, "grad_norm": 9.946159362792969, "learning_rate": 9.536363636363637e-06, "loss": -0.069, "num_tokens": 324984.0, "reward": 0.666408896446228, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7901514172554016, "reward_meter_std": 0.3216681182384491, "reward_std": 0.4116549491882324, "reward_total_composite_mean": 0.666408896446228, "reward_total_composite_std": 0.4116549491882324, "reward_total_mean": 0.666408896446228, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7901514172554016, "rewards/meter/std": 0.3216681182384491, "rewards/total_composite/mean": 0.666408896446228, "rewards/total_composite/std": 0.4116549491882324, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0114688873291016, "sampling/importance_sampling_ratio/min": 0.1386878341436386, "sampling/sampling_logp_difference/max": 1.975529670715332, "sampling/sampling_logp_difference/mean": 0.20986607670783997, "step": 154 }, { "clip_ratio/high_max": 0.08849271759390831, "clip_ratio/high_mean": 0.08849271759390831, "clip_ratio/low_mean": 0.06773262470960617, "clip_ratio/low_min": 0.06773262470960617, "clip_ratio/region_mean": 0.15622534230351448, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 235.5, "completions/mean_terminated_length": 235.5, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 2.7660705894231796, "epoch": 0.005985480383070745, "frac_reward_zero_std": 0.0, "grad_norm": 5.41533899307251, "learning_rate": 9.533333333333334e-06, "loss": -0.0834, "num_tokens": 328524.0, "reward": 0.6477268934249878, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_meter_mean": 0.7103838920593262, "reward_meter_std": 0.29926469922065735, "reward_std": 0.3904498219490051, "reward_total_composite_mean": 0.6477268934249878, "reward_total_composite_std": 0.3904498517513275, "reward_total_mean": 0.6477268934249878, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/meter/mean": 0.7103838920593262, "rewards/meter/std": 0.29926469922065735, "rewards/total_composite/mean": 0.6477268934249878, "rewards/total_composite/std": 0.3904498517513275, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0317730903625488, "sampling/importance_sampling_ratio/min": 0.1159641370177269, "sampling/sampling_logp_difference/max": 2.1544742584228516, "sampling/sampling_logp_difference/mean": 0.19282367825508118, "step": 155 }, { "clip_ratio/high_max": 0.08997475728392601, "clip_ratio/high_mean": 0.08997475728392601, "clip_ratio/low_mean": 0.05361151322722435, "clip_ratio/low_min": 0.05361151322722435, "clip_ratio/region_mean": 0.14358627051115036, "completions/clipped_ratio": 0.0, "completions/max_length": 117.0, "completions/max_terminated_length": 117.0, "completions/mean_length": 105.625, "completions/mean_terminated_length": 105.625, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 1.4543365985155106, "epoch": 0.006024096385542169, "frac_reward_zero_std": 0.0, "grad_norm": 8.701234817504883, "learning_rate": 9.530303030303031e-06, "loss": 0.0523, "num_tokens": 330785.0, "reward": 0.5999155044555664, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.6397774815559387, "reward_meter_std": 0.31731224060058594, "reward_std": 0.2907305359840393, "reward_total_composite_mean": 0.5999155044555664, "reward_total_composite_std": 0.2907305359840393, "reward_total_mean": 0.5999155044555664, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.6397774815559387, "rewards/meter/std": 0.31731224060058594, "rewards/total_composite/mean": 0.5999155044555664, "rewards/total_composite/std": 0.2907305359840393, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0126723051071167, "sampling/importance_sampling_ratio/min": 0.18579503893852234, "sampling/sampling_logp_difference/max": 1.6831111907958984, "sampling/sampling_logp_difference/mean": 0.16831091046333313, "step": 156 }, { "clip_ratio/high_max": 0.14061870239675045, "clip_ratio/high_mean": 0.14061870239675045, "clip_ratio/low_mean": 0.026209676638245583, "clip_ratio/low_min": 0.026209676638245583, "clip_ratio/region_mean": 0.16682837903499603, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 62.625, "completions/mean_terminated_length": 62.625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 2.5809740722179413, "epoch": 0.006062712388013593, "frac_reward_zero_std": 0.0, "grad_norm": 9.539356231689453, "learning_rate": 9.527272727272729e-06, "loss": 0.0172, "num_tokens": 332526.0, "reward": 0.8717015385627747, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8717015385627747, "reward_meter_std": 0.322393000125885, "reward_std": 0.322393000125885, "reward_total_composite_mean": 0.8717015385627747, "reward_total_composite_std": 0.322393000125885, "reward_total_mean": 0.8717015385627747, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8717015385627747, "rewards/meter/std": 0.322393000125885, "rewards/total_composite/mean": 0.8717015385627747, "rewards/total_composite/std": 0.322393000125885, "sampling/importance_sampling_ratio/max": 1.7814452648162842, "sampling/importance_sampling_ratio/mean": 1.0461691617965698, "sampling/importance_sampling_ratio/min": 0.20349915325641632, "sampling/sampling_logp_difference/max": 1.5920934677124023, "sampling/sampling_logp_difference/mean": 0.18296034634113312, "step": 157 }, { "clip_ratio/high_max": 0.11192020168527961, "clip_ratio/high_mean": 0.11192020168527961, "clip_ratio/low_mean": 0.03427810175344348, "clip_ratio/low_min": 0.03427810175344348, "clip_ratio/region_mean": 0.1461983034387231, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 32.375, "completions/mean_terminated_length": 32.375, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 1.62221197783947, "epoch": 0.006101328390485017, "frac_reward_zero_std": 0.0, "grad_norm": 13.795117378234863, "learning_rate": 9.524242424242424e-06, "loss": 0.1023, "num_tokens": 334145.0, "reward": 0.86896812915802, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.86896812915802, "reward_meter_std": 0.2910270094871521, "reward_std": 0.2910270094871521, "reward_total_composite_mean": 0.86896812915802, "reward_total_composite_std": 0.2910270094871521, "reward_total_mean": 0.86896812915802, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.86896812915802, "rewards/meter/std": 0.2910270094871521, "rewards/total_composite/mean": 0.86896812915802, "rewards/total_composite/std": 0.2910270094871521, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0380147695541382, "sampling/importance_sampling_ratio/min": 0.24247469007968903, "sampling/sampling_logp_difference/max": 1.4168579578399658, "sampling/sampling_logp_difference/mean": 0.19816143810749054, "step": 158 }, { "clip_ratio/high_max": 0.13957421202212572, "clip_ratio/high_mean": 0.13957421202212572, "clip_ratio/low_mean": 0.043226663023233414, "clip_ratio/low_min": 0.043226663023233414, "clip_ratio/region_mean": 0.18280087504535913, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 56.5, "completions/mean_terminated_length": 56.5, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 2.072466492652893, "epoch": 0.006139944392956441, "frac_reward_zero_std": 0.0, "grad_norm": 11.608582496643066, "learning_rate": 9.521212121212121e-06, "loss": 0.0781, "num_tokens": 335861.0, "reward": 0.6755526065826416, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6755526065826416, "reward_meter_std": 0.40379220247268677, "reward_std": 0.40379223227500916, "reward_total_composite_mean": 0.6755526065826416, "reward_total_composite_std": 0.40379220247268677, "reward_total_mean": 0.6755526065826416, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6755526065826416, "rewards/meter/std": 0.40379220247268677, "rewards/total_composite/mean": 0.6755526065826416, "rewards/total_composite/std": 0.40379220247268677, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0527642965316772, "sampling/importance_sampling_ratio/min": 0.32885676622390747, "sampling/sampling_logp_difference/max": 1.1121330261230469, "sampling/sampling_logp_difference/mean": 0.1799980252981186, "step": 159 }, { "clip_ratio/high_max": 0.08933841064572334, "clip_ratio/high_mean": 0.08933841064572334, "clip_ratio/low_mean": 0.07780237961560488, "clip_ratio/low_min": 0.07780237961560488, "clip_ratio/region_mean": 0.16714079026132822, "completions/clipped_ratio": 0.0, "completions/max_length": 113.0, "completions/max_terminated_length": 113.0, "completions/mean_length": 81.625, "completions/mean_terminated_length": 81.625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 2.947203129529953, "epoch": 0.006178560395427865, "frac_reward_zero_std": 0.0, "grad_norm": 9.607455253601074, "learning_rate": 9.518181818181819e-06, "loss": -0.1254, "num_tokens": 337794.0, "reward": 0.5209656357765198, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5209656357765198, "reward_meter_std": 0.3587253987789154, "reward_std": 0.358725368976593, "reward_total_composite_mean": 0.5209656357765198, "reward_total_composite_std": 0.3587253987789154, "reward_total_mean": 0.5209656357765198, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5209656357765198, "rewards/meter/std": 0.3587253987789154, "rewards/total_composite/mean": 0.5209656357765198, "rewards/total_composite/std": 0.3587253987789154, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.049707293510437, "sampling/importance_sampling_ratio/min": 0.35801321268081665, "sampling/sampling_logp_difference/max": 1.0271854400634766, "sampling/sampling_logp_difference/mean": 0.20390784740447998, "step": 160 }, { "clip_ratio/high_max": 0.15176175348460674, "clip_ratio/high_mean": 0.15176175348460674, "clip_ratio/low_mean": 0.0405045822262764, "clip_ratio/low_min": 0.0405045822262764, "clip_ratio/region_mean": 0.19226633571088314, "completions/clipped_ratio": 0.0, "completions/max_length": 206.0, "completions/max_terminated_length": 206.0, "completions/mean_length": 173.5, "completions/mean_terminated_length": 173.5, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 3.2587929368019104, "epoch": 0.0062171763978992895, "frac_reward_zero_std": 0.0, "grad_norm": 6.381593704223633, "learning_rate": 9.515151515151516e-06, "loss": -0.0557, "num_tokens": 340718.0, "reward": 0.8628791570663452, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.8734483122825623, "reward_meter_std": 0.19228114187717438, "reward_std": 0.2208014875650406, "reward_total_composite_mean": 0.8628791570663452, "reward_total_composite_std": 0.2208014875650406, "reward_total_mean": 0.8628791570663452, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.8734483122825623, "rewards/meter/std": 0.19228114187717438, "rewards/total_composite/mean": 0.8628791570663452, "rewards/total_composite/std": 0.2208014875650406, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.044055700302124, "sampling/importance_sampling_ratio/min": 0.3062398135662079, "sampling/sampling_logp_difference/max": 1.1833868026733398, "sampling/sampling_logp_difference/mean": 0.20080114901065826, "step": 161 }, { "clip_ratio/high_max": 0.15742158330976963, "clip_ratio/high_mean": 0.15742158330976963, "clip_ratio/low_mean": 0.07329358533024788, "clip_ratio/low_min": 0.07329358533024788, "clip_ratio/region_mean": 0.2307151686400175, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 76.625, "completions/mean_terminated_length": 76.625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 2.3186896294355392, "epoch": 0.006255792400370714, "frac_reward_zero_std": 0.0, "grad_norm": 12.352027893066406, "learning_rate": 9.512121212121213e-06, "loss": -0.0145, "num_tokens": 342611.0, "reward": 0.7090976238250732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7090976238250732, "reward_meter_std": 0.351256400346756, "reward_std": 0.3512563705444336, "reward_total_composite_mean": 0.7090976238250732, "reward_total_composite_std": 0.351256400346756, "reward_total_mean": 0.7090976238250732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7090976238250732, "rewards/meter/std": 0.351256400346756, "rewards/total_composite/mean": 0.7090976238250732, "rewards/total_composite/std": 0.351256400346756, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0393997430801392, "sampling/importance_sampling_ratio/min": 0.2703118920326233, "sampling/sampling_logp_difference/max": 1.341689109802246, "sampling/sampling_logp_difference/mean": 0.2120855152606964, "step": 162 }, { "clip_ratio/high_max": 0.09156403131783009, "clip_ratio/high_mean": 0.09156403131783009, "clip_ratio/low_mean": 0.04785826615989208, "clip_ratio/low_min": 0.04785826615989208, "clip_ratio/region_mean": 0.13942229747772217, "completions/clipped_ratio": 0.0, "completions/max_length": 504.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 451.0, "completions/mean_terminated_length": 451.0, "completions/min_length": 392.0, "completions/min_terminated_length": 392.0, "entropy": 3.1186185479164124, "epoch": 0.006294408402842138, "frac_reward_zero_std": 0.0, "grad_norm": 3.614668846130371, "learning_rate": 9.50909090909091e-06, "loss": 0.0554, "num_tokens": 347883.0, "reward": 0.5725052356719971, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8181818127632141, "reward_count_adherence_std": 0.06872081756591797, "reward_meter_mean": 0.8174892663955688, "reward_meter_std": 0.28463214635849, "reward_std": 0.3688006103038788, "reward_total_composite_mean": 0.5725052356719971, "reward_total_composite_std": 0.36880064010620117, "reward_total_mean": 0.5725052356719971, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8181818127632141, "rewards/count_adherence/std": 0.06872081756591797, "rewards/meter/mean": 0.8174892663955688, "rewards/meter/std": 0.28463214635849, "rewards/total_composite/mean": 0.5725052356719971, "rewards/total_composite/std": 0.36880064010620117, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.045919418334961, "sampling/importance_sampling_ratio/min": 0.09532421082258224, "sampling/sampling_logp_difference/max": 2.3504714965820312, "sampling/sampling_logp_difference/mean": 0.1946493536233902, "step": 163 }, { "clip_ratio/high_max": 0.09482496604323387, "clip_ratio/high_mean": 0.09482496604323387, "clip_ratio/low_mean": 0.11853201128542423, "clip_ratio/low_min": 0.11853201128542423, "clip_ratio/region_mean": 0.2133569773286581, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 3.003392845392227, "epoch": 0.006333024405313562, "frac_reward_zero_std": 0.0, "grad_norm": 10.339271545410156, "learning_rate": 9.506060606060606e-06, "loss": 0.0587, "num_tokens": 349747.0, "reward": 0.47777289152145386, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.47777289152145386, "reward_meter_std": 0.44221097230911255, "reward_std": 0.44221097230911255, "reward_total_composite_mean": 0.47777289152145386, "reward_total_composite_std": 0.44221097230911255, "reward_total_mean": 0.47777289152145386, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.47777289152145386, "rewards/meter/std": 0.44221097230911255, "rewards/total_composite/mean": 0.47777289152145386, "rewards/total_composite/std": 0.44221097230911255, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0429495573043823, "sampling/importance_sampling_ratio/min": 0.2992764711380005, "sampling/sampling_logp_difference/max": 1.2063875198364258, "sampling/sampling_logp_difference/mean": 0.2009221911430359, "step": 164 }, { "clip_ratio/high_max": 0.07877860497683287, "clip_ratio/high_mean": 0.07877860497683287, "clip_ratio/low_mean": 0.08252524584531784, "clip_ratio/low_min": 0.08252524584531784, "clip_ratio/region_mean": 0.1613038508221507, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 61.5, "completions/mean_terminated_length": 61.5, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 2.010311871767044, "epoch": 0.006371640407784986, "frac_reward_zero_std": 0.0, "grad_norm": 13.097195625305176, "learning_rate": 9.503030303030303e-06, "loss": 0.0353, "num_tokens": 351519.0, "reward": 0.5606287121772766, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5606287121772766, "reward_meter_std": 0.3949960768222809, "reward_std": 0.39499610662460327, "reward_total_composite_mean": 0.5606287121772766, "reward_total_composite_std": 0.3949960768222809, "reward_total_mean": 0.5606287121772766, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5606287121772766, "rewards/meter/std": 0.3949960768222809, "rewards/total_composite/mean": 0.5606287121772766, "rewards/total_composite/std": 0.3949960768222809, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0260679721832275, "sampling/importance_sampling_ratio/min": 0.20480231940746307, "sampling/sampling_logp_difference/max": 1.585710048675537, "sampling/sampling_logp_difference/mean": 0.18365828692913055, "step": 165 }, { "clip_ratio/high_max": 0.13404671289026737, "clip_ratio/high_mean": 0.13404671289026737, "clip_ratio/low_mean": 0.0659086387604475, "clip_ratio/low_min": 0.0659086387604475, "clip_ratio/region_mean": 0.19995535165071487, "completions/clipped_ratio": 0.0, "completions/max_length": 159.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 123.625, "completions/mean_terminated_length": 123.625, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 3.55351784825325, "epoch": 0.00641025641025641, "frac_reward_zero_std": 0.0, "grad_norm": 7.968050956726074, "learning_rate": 9.5e-06, "loss": -0.0261, "num_tokens": 353796.0, "reward": 0.35550376772880554, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.37855178117752075, "reward_meter_std": 0.23558726906776428, "reward_std": 0.26453739404678345, "reward_total_composite_mean": 0.35550376772880554, "reward_total_composite_std": 0.26453739404678345, "reward_total_mean": 0.35550376772880554, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.37855178117752075, "rewards/meter/std": 0.23558726906776428, "rewards/total_composite/mean": 0.35550376772880554, "rewards/total_composite/std": 0.26453739404678345, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0575261116027832, "sampling/importance_sampling_ratio/min": 0.24354171752929688, "sampling/sampling_logp_difference/max": 1.4124670028686523, "sampling/sampling_logp_difference/mean": 0.23242639005184174, "step": 166 }, { "clip_ratio/high_max": 0.11530855856835842, "clip_ratio/high_mean": 0.11530855856835842, "clip_ratio/low_mean": 0.06835224945098162, "clip_ratio/low_min": 0.06835224945098162, "clip_ratio/region_mean": 0.18366080801934004, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 53.625, "completions/mean_terminated_length": 53.625, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 2.5232774317264557, "epoch": 0.006448872412727834, "frac_reward_zero_std": 0.0, "grad_norm": 12.225397109985352, "learning_rate": 9.496969696969698e-06, "loss": 0.128, "num_tokens": 355489.0, "reward": 0.541317880153656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.541317880153656, "reward_meter_std": 0.35997992753982544, "reward_std": 0.35997989773750305, "reward_total_composite_mean": 0.541317880153656, "reward_total_composite_std": 0.35997992753982544, "reward_total_mean": 0.541317880153656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.541317880153656, "rewards/meter/std": 0.35997992753982544, "rewards/total_composite/mean": 0.541317880153656, "rewards/total_composite/std": 0.35997992753982544, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0362927913665771, "sampling/importance_sampling_ratio/min": 0.2710648775100708, "sampling/sampling_logp_difference/max": 1.56561279296875, "sampling/sampling_logp_difference/mean": 0.19463400542736053, "step": 167 }, { "clip_ratio/high_max": 0.08096018619835377, "clip_ratio/high_mean": 0.08096018619835377, "clip_ratio/low_mean": 0.08560106623917818, "clip_ratio/low_min": 0.08560106623917818, "clip_ratio/region_mean": 0.16656125243753195, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 72.375, "completions/mean_terminated_length": 72.375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 2.740328684449196, "epoch": 0.006487488415199258, "frac_reward_zero_std": 0.0, "grad_norm": 9.189430236816406, "learning_rate": 9.493939393939395e-06, "loss": 0.0174, "num_tokens": 357276.0, "reward": 0.521911084651947, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.521911084651947, "reward_meter_std": 0.40068677067756653, "reward_std": 0.40068677067756653, "reward_total_composite_mean": 0.521911084651947, "reward_total_composite_std": 0.40068677067756653, "reward_total_mean": 0.521911084651947, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.521911084651947, "rewards/meter/std": 0.40068677067756653, "rewards/total_composite/mean": 0.521911084651947, "rewards/total_composite/std": 0.40068677067756653, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0414764881134033, "sampling/importance_sampling_ratio/min": 0.2703523635864258, "sampling/sampling_logp_difference/max": 1.3080291748046875, "sampling/sampling_logp_difference/mean": 0.19903360307216644, "step": 168 }, { "clip_ratio/high_max": 0.12107311747968197, "clip_ratio/high_mean": 0.12107311747968197, "clip_ratio/low_mean": 0.08651185780763626, "clip_ratio/low_min": 0.08651185780763626, "clip_ratio/region_mean": 0.20758497528731823, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 47.875, "completions/mean_terminated_length": 47.875, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 3.0451967269182205, "epoch": 0.006526104417670682, "frac_reward_zero_std": 0.0, "grad_norm": 13.522438049316406, "learning_rate": 9.490909090909092e-06, "loss": 0.2465, "num_tokens": 358907.0, "reward": 0.4249013066291809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4249013066291809, "reward_meter_std": 0.4389844834804535, "reward_std": 0.4389844834804535, "reward_total_composite_mean": 0.4249013066291809, "reward_total_composite_std": 0.4389844834804535, "reward_total_mean": 0.4249013066291809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4249013066291809, "rewards/meter/std": 0.4389844834804535, "rewards/total_composite/mean": 0.4249013066291809, "rewards/total_composite/std": 0.4389844834804535, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0431842803955078, "sampling/importance_sampling_ratio/min": 0.3039137125015259, "sampling/sampling_logp_difference/max": 1.1910114288330078, "sampling/sampling_logp_difference/mean": 0.2200341522693634, "step": 169 }, { "clip_ratio/high_max": 0.08509290590882301, "clip_ratio/high_mean": 0.08509290590882301, "clip_ratio/low_mean": 0.0997670404613018, "clip_ratio/low_min": 0.0997670404613018, "clip_ratio/region_mean": 0.18485994637012482, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 33.375, "completions/mean_terminated_length": 33.375, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 3.248675227165222, "epoch": 0.006564720420142107, "frac_reward_zero_std": 0.0, "grad_norm": 16.293785095214844, "learning_rate": 9.487878787878788e-06, "loss": 0.0794, "num_tokens": 360430.0, "reward": 0.33747631311416626, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.46223390102386475, "reward_meter_std": 0.4552871286869049, "reward_std": 0.423090398311615, "reward_total_composite_mean": 0.33747631311416626, "reward_total_composite_std": 0.423090398311615, "reward_total_mean": 0.33747631311416626, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.46223390102386475, "rewards/meter/std": 0.4552871286869049, "rewards/total_composite/mean": 0.33747631311416626, "rewards/total_composite/std": 0.423090398311615, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0465396642684937, "sampling/importance_sampling_ratio/min": 0.3476763367652893, "sampling/sampling_logp_difference/max": 1.056483268737793, "sampling/sampling_logp_difference/mean": 0.21340948343276978, "step": 170 }, { "clip_ratio/high_max": 0.09890040010213852, "clip_ratio/high_mean": 0.09890040010213852, "clip_ratio/low_mean": 0.045382389798760414, "clip_ratio/low_min": 0.045382389798760414, "clip_ratio/region_mean": 0.14428278990089893, "completions/clipped_ratio": 0.0, "completions/max_length": 393.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 372.5, "completions/mean_terminated_length": 372.5, "completions/min_length": 346.0, "completions/min_terminated_length": 346.0, "entropy": 2.7181632220745087, "epoch": 0.006603336422613531, "frac_reward_zero_std": 0.0, "grad_norm": 3.9159412384033203, "learning_rate": 9.484848484848485e-06, "loss": 0.0339, "num_tokens": 364930.0, "reward": 0.6796014308929443, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.05143444612622261, "reward_meter_mean": 0.8391759395599365, "reward_meter_std": 0.14736339449882507, "reward_std": 0.3177638053894043, "reward_total_composite_mean": 0.6796014308929443, "reward_total_composite_std": 0.3177638053894043, "reward_total_mean": 0.6796014308929443, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.05143444612622261, "rewards/meter/mean": 0.8391759395599365, "rewards/meter/std": 0.14736339449882507, "rewards/total_composite/mean": 0.6796014308929443, "rewards/total_composite/std": 0.3177638053894043, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0409563779830933, "sampling/importance_sampling_ratio/min": 0.2202359437942505, "sampling/sampling_logp_difference/max": 1.5130558013916016, "sampling/sampling_logp_difference/mean": 0.18074527382850647, "step": 171 }, { "clip_ratio/high_max": 0.04081328213214874, "clip_ratio/high_mean": 0.04081328213214874, "clip_ratio/low_mean": 0.13063165172934532, "clip_ratio/low_min": 0.13063165172934532, "clip_ratio/region_mean": 0.17144493386149406, "completions/clipped_ratio": 0.0, "completions/max_length": 141.0, "completions/max_terminated_length": 141.0, "completions/mean_length": 93.5, "completions/mean_terminated_length": 93.5, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 2.3088208436965942, "epoch": 0.0066419524250849555, "frac_reward_zero_std": 0.0, "grad_norm": 10.551653861999512, "learning_rate": 9.481818181818182e-06, "loss": -0.0949, "num_tokens": 367070.0, "reward": 0.3134363889694214, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3134363889694214, "reward_meter_std": 0.3427402675151825, "reward_std": 0.3427402675151825, "reward_total_composite_mean": 0.3134363889694214, "reward_total_composite_std": 0.3427402675151825, "reward_total_mean": 0.3134363889694214, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3134363889694214, "rewards/meter/std": 0.3427402675151825, "rewards/total_composite/mean": 0.3134363889694214, "rewards/total_composite/std": 0.3427402675151825, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0271856784820557, "sampling/importance_sampling_ratio/min": 0.2899644374847412, "sampling/sampling_logp_difference/max": 1.237997055053711, "sampling/sampling_logp_difference/mean": 0.209889754652977, "step": 172 }, { "clip_ratio/high_max": 0.06190796010196209, "clip_ratio/high_mean": 0.06190796010196209, "clip_ratio/low_mean": 0.05308797210454941, "clip_ratio/low_min": 0.05308797210454941, "clip_ratio/region_mean": 0.1149959322065115, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 504.0, "completions/mean_length": 451.0, "completions/mean_terminated_length": 442.2857360839844, "completions/min_length": 316.0, "completions/min_terminated_length": 316.0, "entropy": 2.9574913382530212, "epoch": 0.00668056842755638, "frac_reward_zero_std": 0.0, "grad_norm": 3.27034068107605, "learning_rate": 9.47878787878788e-06, "loss": 0.1304, "num_tokens": 371854.0, "reward": 0.43490633368492126, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.9204545617103577, "reward_count_adherence_std": 0.058260902762413025, "reward_meter_mean": 0.6058165431022644, "reward_meter_std": 0.27364325523376465, "reward_std": 0.3389948904514313, "reward_total_composite_mean": 0.43490633368492126, "reward_total_composite_std": 0.33899492025375366, "reward_total_mean": 0.43490633368492126, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.9204545617103577, "rewards/count_adherence/std": 0.058260902762413025, "rewards/meter/mean": 0.6058165431022644, "rewards/meter/std": 0.27364325523376465, "rewards/total_composite/mean": 0.43490633368492126, "rewards/total_composite/std": 0.33899492025375366, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0502684116363525, "sampling/importance_sampling_ratio/min": 0.22345705330371857, "sampling/sampling_logp_difference/max": 1.4985361099243164, "sampling/sampling_logp_difference/mean": 0.19972197711467743, "step": 173 }, { "clip_ratio/high_max": 0.13539652153849602, "clip_ratio/high_mean": 0.13539652153849602, "clip_ratio/low_mean": 0.05789176933467388, "clip_ratio/low_min": 0.05789176933467388, "clip_ratio/region_mean": 0.1932882908731699, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 55.875, "completions/mean_terminated_length": 55.875, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 2.167596936225891, "epoch": 0.006719184430027804, "frac_reward_zero_std": 0.0, "grad_norm": 11.655610084533691, "learning_rate": 9.475757575757577e-06, "loss": 0.0781, "num_tokens": 373685.0, "reward": 0.7013630867004395, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7013630867004395, "reward_meter_std": 0.37237003445625305, "reward_std": 0.37237003445625305, "reward_total_composite_mean": 0.7013630867004395, "reward_total_composite_std": 0.37237003445625305, "reward_total_mean": 0.7013630867004395, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7013630867004395, "rewards/meter/std": 0.37237003445625305, "rewards/total_composite/mean": 0.7013630867004395, "rewards/total_composite/std": 0.37237003445625305, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0189765691757202, "sampling/importance_sampling_ratio/min": 0.16070450842380524, "sampling/sampling_logp_difference/max": 1.8281879425048828, "sampling/sampling_logp_difference/mean": 0.19981878995895386, "step": 174 }, { "clip_ratio/high_max": 0.1139191659167409, "clip_ratio/high_mean": 0.1139191659167409, "clip_ratio/low_mean": 0.05846922844648361, "clip_ratio/low_min": 0.05846922844648361, "clip_ratio/region_mean": 0.1723883943632245, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 99.25, "completions/mean_terminated_length": 99.25, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 3.0380934178829193, "epoch": 0.006757800432499228, "frac_reward_zero_std": 0.0, "grad_norm": 8.652692794799805, "learning_rate": 9.472727272727274e-06, "loss": -0.0845, "num_tokens": 375783.0, "reward": 0.8306900858879089, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8306900858879089, "reward_meter_std": 0.2705877721309662, "reward_std": 0.2705877721309662, "reward_total_composite_mean": 0.8306900858879089, "reward_total_composite_std": 0.2705877721309662, "reward_total_mean": 0.8306900858879089, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8306900858879089, "rewards/meter/std": 0.2705877721309662, "rewards/total_composite/mean": 0.8306900858879089, "rewards/total_composite/std": 0.2705877721309662, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0498780012130737, "sampling/importance_sampling_ratio/min": 0.1981683075428009, "sampling/sampling_logp_difference/max": 1.618638515472412, "sampling/sampling_logp_difference/mean": 0.20954445004463196, "step": 175 }, { "clip_ratio/high_max": 0.10677518881857395, "clip_ratio/high_mean": 0.10677518881857395, "clip_ratio/low_mean": 0.07028611842542887, "clip_ratio/low_min": 0.07028611842542887, "clip_ratio/region_mean": 0.17706130724400282, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 55.75, "completions/mean_terminated_length": 55.75, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 1.1572269052267075, "epoch": 0.006796416434970652, "frac_reward_zero_std": 0.0, "grad_norm": 18.11098861694336, "learning_rate": 9.469696969696971e-06, "loss": 0.0503, "num_tokens": 377613.0, "reward": 0.5684157609939575, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5684157609939575, "reward_meter_std": 0.3562447726726532, "reward_std": 0.3562447726726532, "reward_total_composite_mean": 0.5684157609939575, "reward_total_composite_std": 0.3562447726726532, "reward_total_mean": 0.5684157609939575, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5684157609939575, "rewards/meter/std": 0.3562447726726532, "rewards/total_composite/mean": 0.5684157609939575, "rewards/total_composite/std": 0.3562447726726532, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9817885756492615, "sampling/importance_sampling_ratio/min": 0.15001113712787628, "sampling/sampling_logp_difference/max": 1.8970457315444946, "sampling/sampling_logp_difference/mean": 0.18087293207645416, "step": 176 }, { "clip_ratio/high_max": 0.10677864961326122, "clip_ratio/high_mean": 0.10677864961326122, "clip_ratio/low_mean": 0.07266573421657085, "clip_ratio/low_min": 0.07266573421657085, "clip_ratio/region_mean": 0.17944438382983208, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 65.625, "completions/mean_terminated_length": 65.625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 2.025869235396385, "epoch": 0.006835032437442076, "frac_reward_zero_std": 0.0, "grad_norm": 11.097681999206543, "learning_rate": 9.466666666666667e-06, "loss": 0.0261, "num_tokens": 379410.0, "reward": 0.6728242039680481, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6728242039680481, "reward_meter_std": 0.35246020555496216, "reward_std": 0.35246017575263977, "reward_total_composite_mean": 0.6728242039680481, "reward_total_composite_std": 0.35246020555496216, "reward_total_mean": 0.6728242039680481, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6728242039680481, "rewards/meter/std": 0.35246020555496216, "rewards/total_composite/mean": 0.6728242039680481, "rewards/total_composite/std": 0.35246020555496216, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0345712900161743, "sampling/importance_sampling_ratio/min": 0.21794182062149048, "sampling/sampling_logp_difference/max": 1.5235271453857422, "sampling/sampling_logp_difference/mean": 0.1819072812795639, "step": 177 }, { "clip_ratio/high_max": 0.053513072431087494, "clip_ratio/high_mean": 0.053513072431087494, "clip_ratio/low_mean": 0.1264551393687725, "clip_ratio/low_min": 0.1264551393687725, "clip_ratio/region_mean": 0.17996821179986, "completions/clipped_ratio": 0.0, "completions/max_length": 112.0, "completions/max_terminated_length": 112.0, "completions/mean_length": 95.375, "completions/mean_terminated_length": 95.375, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 2.3764619678258896, "epoch": 0.0068736484399135, "frac_reward_zero_std": 0.0, "grad_norm": 8.50960636138916, "learning_rate": 9.463636363636364e-06, "loss": 0.1003, "num_tokens": 381389.0, "reward": 0.2792768180370331, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2792768180370331, "reward_meter_std": 0.3471120297908783, "reward_std": 0.3471120595932007, "reward_total_composite_mean": 0.2792768180370331, "reward_total_composite_std": 0.3471120297908783, "reward_total_mean": 0.2792768180370331, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2792768180370331, "rewards/meter/std": 0.3471120297908783, "rewards/total_composite/mean": 0.2792768180370331, "rewards/total_composite/std": 0.3471120297908783, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0582326650619507, "sampling/importance_sampling_ratio/min": 0.331091970205307, "sampling/sampling_logp_difference/max": 1.1053590774536133, "sampling/sampling_logp_difference/mean": 0.18758107721805573, "step": 178 }, { "clip_ratio/high_max": 0.15819299593567848, "clip_ratio/high_mean": 0.15819299593567848, "clip_ratio/low_mean": 0.05450693517923355, "clip_ratio/low_min": 0.05450693517923355, "clip_ratio/region_mean": 0.21269993111491203, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 39.0, "completions/mean_terminated_length": 39.0, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 2.807357758283615, "epoch": 0.006912264442384924, "frac_reward_zero_std": 0.0, "grad_norm": 13.35677433013916, "learning_rate": 9.460606060606061e-06, "loss": 0.0501, "num_tokens": 382901.0, "reward": 0.742904782295227, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.742904782295227, "reward_meter_std": 0.4480074644088745, "reward_std": 0.4480074346065521, "reward_total_composite_mean": 0.742904782295227, "reward_total_composite_std": 0.4480074644088745, "reward_total_mean": 0.742904782295227, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.742904782295227, "rewards/meter/std": 0.4480074644088745, "rewards/total_composite/mean": 0.742904782295227, "rewards/total_composite/std": 0.4480074644088745, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0228500366210938, "sampling/importance_sampling_ratio/min": 0.21225318312644958, "sampling/sampling_logp_difference/max": 1.5499753952026367, "sampling/sampling_logp_difference/mean": 0.2040482759475708, "step": 179 }, { "clip_ratio/high_max": 0.0775204561650753, "clip_ratio/high_mean": 0.0775204561650753, "clip_ratio/low_mean": 0.10372502729296684, "clip_ratio/low_min": 0.10372502729296684, "clip_ratio/region_mean": 0.18124548345804214, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 118.875, "completions/mean_terminated_length": 118.875, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 2.9363907277584076, "epoch": 0.006950880444856348, "frac_reward_zero_std": 0.0, "grad_norm": 7.980728626251221, "learning_rate": 9.457575757575759e-06, "loss": -0.0462, "num_tokens": 385212.0, "reward": 0.4976061284542084, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.6352285146713257, "reward_meter_std": 0.3728267252445221, "reward_std": 0.40696853399276733, "reward_total_composite_mean": 0.4976061284542084, "reward_total_composite_std": 0.40696853399276733, "reward_total_mean": 0.4976061284542084, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.6352285146713257, "rewards/meter/std": 0.3728267252445221, "rewards/total_composite/mean": 0.4976061284542084, "rewards/total_composite/std": 0.40696853399276733, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0363506078720093, "sampling/importance_sampling_ratio/min": 0.2866778075695038, "sampling/sampling_logp_difference/max": 1.2493963241577148, "sampling/sampling_logp_difference/mean": 0.20611317455768585, "step": 180 }, { "clip_ratio/high_max": 0.11146864108741283, "clip_ratio/high_mean": 0.11146864108741283, "clip_ratio/low_mean": 0.051841133274137974, "clip_ratio/low_min": 0.051841133274137974, "clip_ratio/region_mean": 0.1633097743615508, "completions/clipped_ratio": 0.0, "completions/max_length": 204.0, "completions/max_terminated_length": 204.0, "completions/mean_length": 174.625, "completions/mean_terminated_length": 174.625, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 3.1949117481708527, "epoch": 0.0069894964473277725, "frac_reward_zero_std": 0.0, "grad_norm": 5.923155784606934, "learning_rate": 9.454545454545456e-06, "loss": 0.0206, "num_tokens": 388081.0, "reward": 0.5388948917388916, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.6006770133972168, "reward_meter_std": 0.31185153126716614, "reward_std": 0.3470546305179596, "reward_total_composite_mean": 0.5388948917388916, "reward_total_composite_std": 0.3470546007156372, "reward_total_mean": 0.5388948917388916, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.6006770133972168, "rewards/meter/std": 0.31185153126716614, "rewards/total_composite/mean": 0.5388948917388916, "rewards/total_composite/std": 0.3470546007156372, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0360268354415894, "sampling/importance_sampling_ratio/min": 0.25199154019355774, "sampling/sampling_logp_difference/max": 1.3783597946166992, "sampling/sampling_logp_difference/mean": 0.2052806168794632, "step": 181 }, { "clip_ratio/high_max": 0.1037982078269124, "clip_ratio/high_mean": 0.1037982078269124, "clip_ratio/low_mean": 0.06412622984498739, "clip_ratio/low_min": 0.06412622984498739, "clip_ratio/region_mean": 0.1679244376718998, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 84.75, "completions/mean_terminated_length": 84.75, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 2.674738720059395, "epoch": 0.007028112449799197, "frac_reward_zero_std": 0.0, "grad_norm": 8.314396858215332, "learning_rate": 9.451515151515153e-06, "loss": -0.0381, "num_tokens": 390055.0, "reward": 0.8241270780563354, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8241270780563354, "reward_meter_std": 0.21255002915859222, "reward_std": 0.21255002915859222, "reward_total_composite_mean": 0.8241270780563354, "reward_total_composite_std": 0.21255002915859222, "reward_total_mean": 0.8241270780563354, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8241270780563354, "rewards/meter/std": 0.21255002915859222, "rewards/total_composite/mean": 0.8241270780563354, "rewards/total_composite/std": 0.21255002915859222, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0361807346343994, "sampling/importance_sampling_ratio/min": 0.211969792842865, "sampling/sampling_logp_difference/max": 1.5513114929199219, "sampling/sampling_logp_difference/mean": 0.1786930412054062, "step": 182 }, { "clip_ratio/high_max": 0.016119910404086113, "clip_ratio/high_mean": 0.016119910404086113, "clip_ratio/low_mean": 0.031273381784558296, "clip_ratio/low_min": 0.031273381784558296, "clip_ratio/region_mean": 0.04739329218864441, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 486.0, "completions/mean_length": 485.5, "completions/mean_terminated_length": 441.3333435058594, "completions/min_length": 396.0, "completions/min_terminated_length": 396.0, "entropy": 1.6914453506469727, "epoch": 0.007066728452270621, "frac_reward_zero_std": 0.0, "grad_norm": 3.2879745960235596, "learning_rate": 9.448484848484849e-06, "loss": -0.2128, "num_tokens": 393147.0, "reward": 0.023662419989705086, "reward_arabic_clean_mean": 0.125, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.7767857313156128, "reward_count_adherence_std": 0.21407301723957062, "reward_meter_mean": 0.26494020223617554, "reward_meter_std": 0.2154451161623001, "reward_std": 0.06692743301391602, "reward_total_composite_mean": 0.023662419989705086, "reward_total_composite_std": 0.06692743301391602, "reward_total_mean": 0.023662419989705086, "rewards/arabic_clean/mean": 0.125, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.7767857313156128, "rewards/count_adherence/std": 0.21407301723957062, "rewards/meter/mean": 0.26494020223617554, "rewards/meter/std": 0.2154451161623001, "rewards/total_composite/mean": 0.023662419989705086, "rewards/total_composite/std": 0.06692743301391602, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0776352882385254, "sampling/importance_sampling_ratio/min": 0.24409537017345428, "sampling/sampling_logp_difference/max": 1.410196304321289, "sampling/sampling_logp_difference/mean": 0.22992289066314697, "step": 183 }, { "clip_ratio/high_max": 0.12044411525130272, "clip_ratio/high_mean": 0.12044411525130272, "clip_ratio/low_mean": 0.06921044550836086, "clip_ratio/low_min": 0.06921044550836086, "clip_ratio/region_mean": 0.18965456075966358, "completions/clipped_ratio": 0.0, "completions/max_length": 199.0, "completions/max_terminated_length": 199.0, "completions/mean_length": 151.375, "completions/mean_terminated_length": 151.375, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 3.480381518602371, "epoch": 0.007105344454742045, "frac_reward_zero_std": 0.0, "grad_norm": 7.487835884094238, "learning_rate": 9.445454545454546e-06, "loss": 0.0335, "num_tokens": 395894.0, "reward": 0.7357177734375, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.7551294565200806, "reward_meter_std": 0.2453630119562149, "reward_std": 0.23533061146736145, "reward_total_composite_mean": 0.7357177734375, "reward_total_composite_std": 0.23533061146736145, "reward_total_mean": 0.7357177734375, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.7551294565200806, "rewards/meter/std": 0.2453630119562149, "rewards/total_composite/mean": 0.7357177734375, "rewards/total_composite/std": 0.23533061146736145, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0623719692230225, "sampling/importance_sampling_ratio/min": 0.1513015180826187, "sampling/sampling_logp_difference/max": 1.888480544090271, "sampling/sampling_logp_difference/mean": 0.22830148041248322, "step": 184 }, { "clip_ratio/high_max": 0.04156912490725517, "clip_ratio/high_mean": 0.04156912490725517, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.04156912490725517, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 434.0, "completions/mean_length": 485.125, "completions/mean_terminated_length": 404.5, "completions/min_length": 375.0, "completions/min_terminated_length": 375.0, "entropy": 1.180066168308258, "epoch": 0.007143960457213469, "frac_reward_zero_std": 0.0, "grad_norm": 1.9176102876663208, "learning_rate": 9.442424242424243e-06, "loss": -0.2352, "num_tokens": 398431.0, "reward": 0.36697742342948914, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.7647058963775635, "reward_count_adherence_std": 0.2037706971168518, "reward_meter_mean": 0.6249135732650757, "reward_meter_std": 0.15311351418495178, "reward_std": 0.280710369348526, "reward_total_composite_mean": 0.36697742342948914, "reward_total_composite_std": 0.280710369348526, "reward_total_mean": 0.36697742342948914, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.7647058963775635, "rewards/count_adherence/std": 0.2037706971168518, "rewards/meter/mean": 0.6249135732650757, "rewards/meter/std": 0.15311351418495178, "rewards/total_composite/mean": 0.36697742342948914, "rewards/total_composite/std": 0.280710369348526, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0640817880630493, "sampling/importance_sampling_ratio/min": 0.2796296775341034, "sampling/sampling_logp_difference/max": 1.2742891311645508, "sampling/sampling_logp_difference/mean": 0.22530639171600342, "step": 185 }, { "clip_ratio/high_max": 0.08783877268433571, "clip_ratio/high_mean": 0.08783877268433571, "clip_ratio/low_mean": 0.14540163800120354, "clip_ratio/low_min": 0.14540163800120354, "clip_ratio/region_mean": 0.23324041068553925, "completions/clipped_ratio": 0.0, "completions/max_length": 159.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 118.75, "completions/mean_terminated_length": 118.75, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 1.863920621573925, "epoch": 0.007182576459684893, "frac_reward_zero_std": 0.0, "grad_norm": 13.209904670715332, "learning_rate": 9.43939393939394e-06, "loss": -0.0388, "num_tokens": 400845.0, "reward": 0.33415091037750244, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625821024179459, "reward_meter_mean": 0.4079614281654358, "reward_meter_std": 0.2506245970726013, "reward_std": 0.2877470552921295, "reward_total_composite_mean": 0.33415091037750244, "reward_total_composite_std": 0.2877470552921295, "reward_total_mean": 0.33415091037750244, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625821024179459, "rewards/meter/mean": 0.4079614281654358, "rewards/meter/std": 0.2506245970726013, "rewards/total_composite/mean": 0.33415091037750244, "rewards/total_composite/std": 0.2877470552921295, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.018273115158081, "sampling/importance_sampling_ratio/min": 0.05264463275671005, "sampling/sampling_logp_difference/max": 2.9441909790039062, "sampling/sampling_logp_difference/mean": 0.25658828020095825, "step": 186 }, { "clip_ratio/high_max": 0.05676225572824478, "clip_ratio/high_mean": 0.05676225572824478, "clip_ratio/low_mean": 0.08290823642164469, "clip_ratio/low_min": 0.08290823642164469, "clip_ratio/region_mean": 0.13967049214988947, "completions/clipped_ratio": 0.0, "completions/max_length": 175.0, "completions/max_terminated_length": 175.0, "completions/mean_length": 163.0, "completions/mean_terminated_length": 163.0, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 1.991146519780159, "epoch": 0.007221192462156318, "frac_reward_zero_std": 0.0, "grad_norm": 6.4664082527160645, "learning_rate": 9.436363636363636e-06, "loss": 0.0416, "num_tokens": 403557.0, "reward": 0.5632787346839905, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_meter_mean": 0.582764744758606, "reward_meter_std": 0.3457765281200409, "reward_std": 0.3654842972755432, "reward_total_composite_mean": 0.5632787346839905, "reward_total_composite_std": 0.3654843270778656, "reward_total_mean": 0.5632787346839905, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/meter/mean": 0.582764744758606, "rewards/meter/std": 0.3457765281200409, "rewards/total_composite/mean": 0.5632787346839905, "rewards/total_composite/std": 0.3654843270778656, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0383338928222656, "sampling/importance_sampling_ratio/min": 0.2662302553653717, "sampling/sampling_logp_difference/max": 1.3233938217163086, "sampling/sampling_logp_difference/mean": 0.1593349426984787, "step": 187 }, { "clip_ratio/high_max": 0.1269478127360344, "clip_ratio/high_mean": 0.1269478127360344, "clip_ratio/low_mean": 0.04282732866704464, "clip_ratio/low_min": 0.04282732866704464, "clip_ratio/region_mean": 0.16977514140307903, "completions/clipped_ratio": 0.0, "completions/max_length": 113.0, "completions/max_terminated_length": 113.0, "completions/mean_length": 93.625, "completions/mean_terminated_length": 93.625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 1.5476962476968765, "epoch": 0.007259808464627742, "frac_reward_zero_std": 0.0, "grad_norm": 13.624017715454102, "learning_rate": 9.433333333333335e-06, "loss": 0.0815, "num_tokens": 405714.0, "reward": 0.5144075155258179, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.6261414885520935, "reward_meter_std": 0.19565246999263763, "reward_std": 0.2748170793056488, "reward_total_composite_mean": 0.5144075155258179, "reward_total_composite_std": 0.2748170793056488, "reward_total_mean": 0.5144075155258179, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.6261414885520935, "rewards/meter/std": 0.19565246999263763, "rewards/total_composite/mean": 0.5144075155258179, "rewards/total_composite/std": 0.2748170793056488, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.026123285293579, "sampling/importance_sampling_ratio/min": 0.1549825519323349, "sampling/sampling_logp_difference/max": 1.8644428253173828, "sampling/sampling_logp_difference/mean": 0.19976428151130676, "step": 188 }, { "clip_ratio/high_max": 0.07970546465367079, "clip_ratio/high_mean": 0.07970546465367079, "clip_ratio/low_mean": 0.08321618381887674, "clip_ratio/low_min": 0.08321618381887674, "clip_ratio/region_mean": 0.16292164847254753, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 55.875, "completions/mean_terminated_length": 55.875, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 1.7971455752849579, "epoch": 0.007298424467099166, "frac_reward_zero_std": 0.0, "grad_norm": 11.650201797485352, "learning_rate": 9.43030303030303e-06, "loss": 0.0495, "num_tokens": 407329.0, "reward": 0.5849356055259705, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5849356055259705, "reward_meter_std": 0.39748692512512207, "reward_std": 0.39748692512512207, "reward_total_composite_mean": 0.5849356055259705, "reward_total_composite_std": 0.39748692512512207, "reward_total_mean": 0.5849356055259705, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5849356055259705, "rewards/meter/std": 0.39748692512512207, "rewards/total_composite/mean": 0.5849356055259705, "rewards/total_composite/std": 0.39748692512512207, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0523135662078857, "sampling/importance_sampling_ratio/min": 0.25276920199394226, "sampling/sampling_logp_difference/max": 1.3752784729003906, "sampling/sampling_logp_difference/mean": 0.16425088047981262, "step": 189 }, { "clip_ratio/high_max": 0.08876706939190626, "clip_ratio/high_mean": 0.08876706939190626, "clip_ratio/low_mean": 0.049503629095852375, "clip_ratio/low_min": 0.049503629095852375, "clip_ratio/region_mean": 0.13827069848775864, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 69.75, "completions/mean_terminated_length": 69.75, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 2.6996882259845734, "epoch": 0.00733704046957059, "frac_reward_zero_std": 0.0, "grad_norm": 10.552634239196777, "learning_rate": 9.427272727272728e-06, "loss": 0.1149, "num_tokens": 409327.0, "reward": 0.6193662881851196, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6343156695365906, "reward_meter_std": 0.4757094979286194, "reward_std": 0.4956565201282501, "reward_total_composite_mean": 0.6193662881851196, "reward_total_composite_std": 0.4956565201282501, "reward_total_mean": 0.6193662881851196, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6343156695365906, "rewards/meter/std": 0.4757094979286194, "rewards/total_composite/mean": 0.6193662881851196, "rewards/total_composite/std": 0.4956565201282501, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0550477504730225, "sampling/importance_sampling_ratio/min": 0.24255730211734772, "sampling/sampling_logp_difference/max": 1.4165172576904297, "sampling/sampling_logp_difference/mean": 0.20032422244548798, "step": 190 }, { "clip_ratio/high_max": 0.10810728743672371, "clip_ratio/high_mean": 0.10810728743672371, "clip_ratio/low_mean": 0.029843377880752087, "clip_ratio/low_min": 0.029843377880752087, "clip_ratio/region_mean": 0.1379506653174758, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 323.875, "completions/mean_terminated_length": 323.875, "completions/min_length": 305.0, "completions/min_terminated_length": 305.0, "entropy": 2.755393832921982, "epoch": 0.007375656472042014, "frac_reward_zero_std": 0.0, "grad_norm": 4.086877822875977, "learning_rate": 9.424242424242425e-06, "loss": 0.0501, "num_tokens": 413518.0, "reward": 0.8315606117248535, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8315606117248535, "reward_meter_std": 0.19411730766296387, "reward_std": 0.19411730766296387, "reward_total_composite_mean": 0.8315606117248535, "reward_total_composite_std": 0.19411730766296387, "reward_total_mean": 0.8315606117248535, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8315606117248535, "rewards/meter/std": 0.19411730766296387, "rewards/total_composite/mean": 0.8315606117248535, "rewards/total_composite/std": 0.19411730766296387, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0462325811386108, "sampling/importance_sampling_ratio/min": 0.19589684903621674, "sampling/sampling_logp_difference/max": 1.630167007446289, "sampling/sampling_logp_difference/mean": 0.1796930879354477, "step": 191 }, { "clip_ratio/high_max": 0.16411421447992325, "clip_ratio/high_mean": 0.16411421447992325, "clip_ratio/low_mean": 0.01315789483487606, "clip_ratio/low_min": 0.01315789483487606, "clip_ratio/region_mean": 0.1772721093147993, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 27.375, "completions/mean_terminated_length": 27.375, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 2.018664211034775, "epoch": 0.0074142724745134385, "frac_reward_zero_std": 0.0, "grad_norm": 18.393901824951172, "learning_rate": 9.421212121212122e-06, "loss": 0.153, "num_tokens": 414921.0, "reward": 0.9494917392730713, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9494917392730713, "reward_meter_std": 0.08201512694358826, "reward_std": 0.08201511204242706, "reward_total_composite_mean": 0.9494917392730713, "reward_total_composite_std": 0.08201512694358826, "reward_total_mean": 0.9494917392730713, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9494917392730713, "rewards/meter/std": 0.08201512694358826, "rewards/total_composite/mean": 0.9494917392730713, "rewards/total_composite/std": 0.08201512694358826, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0469014644622803, "sampling/importance_sampling_ratio/min": 0.22913827002048492, "sampling/sampling_logp_difference/max": 1.4734296798706055, "sampling/sampling_logp_difference/mean": 0.1969519406557083, "step": 192 }, { "clip_ratio/high_max": 0.07883104495704174, "clip_ratio/high_mean": 0.07883104495704174, "clip_ratio/low_mean": 0.07560309767723083, "clip_ratio/low_min": 0.07560309767723083, "clip_ratio/region_mean": 0.15443414263427258, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 95.25, "completions/mean_terminated_length": 95.25, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 2.2031291872262955, "epoch": 0.007452888476984863, "frac_reward_zero_std": 0.0, "grad_norm": 8.817331314086914, "learning_rate": 9.418181818181818e-06, "loss": 0.0298, "num_tokens": 416883.0, "reward": 0.6673023104667664, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6673023104667664, "reward_meter_std": 0.2610948383808136, "reward_std": 0.2610948085784912, "reward_total_composite_mean": 0.6673023104667664, "reward_total_composite_std": 0.2610948383808136, "reward_total_mean": 0.6673023104667664, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6673023104667664, "rewards/meter/std": 0.2610948383808136, "rewards/total_composite/mean": 0.6673023104667664, "rewards/total_composite/std": 0.2610948383808136, "sampling/importance_sampling_ratio/max": 1.9678956270217896, "sampling/importance_sampling_ratio/mean": 1.0500800609588623, "sampling/importance_sampling_ratio/min": 0.21263186633586884, "sampling/sampling_logp_difference/max": 1.5481929779052734, "sampling/sampling_logp_difference/mean": 0.18318572640419006, "step": 193 }, { "clip_ratio/high_max": 0.10805463790893555, "clip_ratio/high_mean": 0.10805463790893555, "clip_ratio/low_mean": 0.09499397501349449, "clip_ratio/low_min": 0.09499397501349449, "clip_ratio/region_mean": 0.20304861292243004, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 2.68495112657547, "epoch": 0.007491504479456287, "frac_reward_zero_std": 0.0, "grad_norm": 9.746423721313477, "learning_rate": 9.415151515151515e-06, "loss": -0.0527, "num_tokens": 418806.0, "reward": 0.5814144611358643, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7032723426818848, "reward_meter_std": 0.39979878067970276, "reward_std": 0.45054084062576294, "reward_total_composite_mean": 0.5814144611358643, "reward_total_composite_std": 0.45054084062576294, "reward_total_mean": 0.5814144611358643, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7032723426818848, "rewards/meter/std": 0.39979878067970276, "rewards/total_composite/mean": 0.5814144611358643, "rewards/total_composite/std": 0.45054084062576294, "sampling/importance_sampling_ratio/max": 1.9423279762268066, "sampling/importance_sampling_ratio/mean": 1.0394792556762695, "sampling/importance_sampling_ratio/min": 0.3289699852466583, "sampling/sampling_logp_difference/max": 1.1117887496948242, "sampling/sampling_logp_difference/mean": 0.1895114779472351, "step": 194 }, { "clip_ratio/high_max": 0.054079256020486355, "clip_ratio/high_mean": 0.054079256020486355, "clip_ratio/low_mean": 0.09933342039585114, "clip_ratio/low_min": 0.09933342039585114, "clip_ratio/region_mean": 0.1534126764163375, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 33.125, "completions/mean_terminated_length": 33.125, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 1.359821267426014, "epoch": 0.007530120481927711, "frac_reward_zero_std": 0.0, "grad_norm": 22.751644134521484, "learning_rate": 9.412121212121212e-06, "loss": 0.1632, "num_tokens": 420399.0, "reward": 0.5200808048248291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5200808048248291, "reward_meter_std": 0.43741583824157715, "reward_std": 0.43741583824157715, "reward_total_composite_mean": 0.5200808048248291, "reward_total_composite_std": 0.43741583824157715, "reward_total_mean": 0.5200808048248291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5200808048248291, "rewards/meter/std": 0.43741583824157715, "rewards/total_composite/mean": 0.5200808048248291, "rewards/total_composite/std": 0.43741583824157715, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.015283226966858, "sampling/importance_sampling_ratio/min": 0.027538040652871132, "sampling/sampling_logp_difference/max": 3.59218692779541, "sampling/sampling_logp_difference/mean": 0.2031438797712326, "step": 195 }, { "clip_ratio/high_max": 0.13002329412847757, "clip_ratio/high_mean": 0.13002329412847757, "clip_ratio/low_mean": 0.04345930367708206, "clip_ratio/low_min": 0.04345930367708206, "clip_ratio/region_mean": 0.17348259780555964, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 61.375, "completions/mean_terminated_length": 61.375, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 2.188107267022133, "epoch": 0.007568736484399135, "frac_reward_zero_std": 0.0, "grad_norm": 14.902490615844727, "learning_rate": 9.40909090909091e-06, "loss": 0.0894, "num_tokens": 422210.0, "reward": 0.6541217565536499, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6579214334487915, "reward_meter_std": 0.34652820229530334, "reward_std": 0.3544676601886749, "reward_total_composite_mean": 0.6541217565536499, "reward_total_composite_std": 0.3544676899909973, "reward_total_mean": 0.6541217565536499, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6579214334487915, "rewards/meter/std": 0.34652820229530334, "rewards/total_composite/mean": 0.6541217565536499, "rewards/total_composite/std": 0.3544676899909973, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0256552696228027, "sampling/importance_sampling_ratio/min": 0.32179901003837585, "sampling/sampling_logp_difference/max": 1.1338281631469727, "sampling/sampling_logp_difference/mean": 0.19266894459724426, "step": 196 }, { "clip_ratio/high_max": 0.1163166556507349, "clip_ratio/high_mean": 0.1163166556507349, "clip_ratio/low_mean": 0.03474697098135948, "clip_ratio/low_min": 0.03474697098135948, "clip_ratio/region_mean": 0.15106362663209438, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 73.875, "completions/mean_terminated_length": 73.875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 2.119014173746109, "epoch": 0.007607352486870559, "frac_reward_zero_std": 0.0, "grad_norm": 8.72506332397461, "learning_rate": 9.406060606060607e-06, "loss": -0.0611, "num_tokens": 423961.0, "reward": 0.7566624879837036, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7866647243499756, "reward_meter_std": 0.33512723445892334, "reward_std": 0.3962303400039673, "reward_total_composite_mean": 0.7566624879837036, "reward_total_composite_std": 0.3962303102016449, "reward_total_mean": 0.7566624879837036, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7866647243499756, "rewards/meter/std": 0.33512723445892334, "rewards/total_composite/mean": 0.7566624879837036, "rewards/total_composite/std": 0.3962303102016449, "sampling/importance_sampling_ratio/max": 1.8759212493896484, "sampling/importance_sampling_ratio/mean": 1.0307003259658813, "sampling/importance_sampling_ratio/min": 0.23423144221305847, "sampling/sampling_logp_difference/max": 1.4514455795288086, "sampling/sampling_logp_difference/mean": 0.16485705971717834, "step": 197 }, { "clip_ratio/high_max": 0.11953294090926647, "clip_ratio/high_mean": 0.11953294090926647, "clip_ratio/low_mean": 0.02056962065398693, "clip_ratio/low_min": 0.02056962065398693, "clip_ratio/region_mean": 0.1401025615632534, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 72.875, "completions/mean_terminated_length": 72.875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 2.166542023420334, "epoch": 0.007645968489341983, "frac_reward_zero_std": 0.0, "grad_norm": 8.083459854125977, "learning_rate": 9.403030303030304e-06, "loss": 0.0372, "num_tokens": 425728.0, "reward": 0.8608720302581787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8608720302581787, "reward_meter_std": 0.3337464928627014, "reward_std": 0.33374646306037903, "reward_total_composite_mean": 0.8608720302581787, "reward_total_composite_std": 0.3337464928627014, "reward_total_mean": 0.8608720302581787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8608720302581787, "rewards/meter/std": 0.3337464928627014, "rewards/total_composite/mean": 0.8608720302581787, "rewards/total_composite/std": 0.3337464928627014, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0441133975982666, "sampling/importance_sampling_ratio/min": 0.27538660168647766, "sampling/sampling_logp_difference/max": 1.2895793914794922, "sampling/sampling_logp_difference/mean": 0.181260883808136, "step": 198 }, { "clip_ratio/high_max": 0.11455865763127804, "clip_ratio/high_mean": 0.11455865763127804, "clip_ratio/low_mean": 0.0449892720207572, "clip_ratio/low_min": 0.0449892720207572, "clip_ratio/region_mean": 0.15954792965203524, "completions/clipped_ratio": 0.0, "completions/max_length": 149.0, "completions/max_terminated_length": 149.0, "completions/mean_length": 140.375, "completions/mean_terminated_length": 140.375, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 2.1282119899988174, "epoch": 0.007684584491813407, "frac_reward_zero_std": 0.0, "grad_norm": 6.834539890289307, "learning_rate": 9.4e-06, "loss": 0.0447, "num_tokens": 428283.0, "reward": 0.8193603754043579, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.8502488136291504, "reward_meter_std": 0.2549903690814972, "reward_std": 0.250792533159256, "reward_total_composite_mean": 0.8193603754043579, "reward_total_composite_std": 0.250792533159256, "reward_total_mean": 0.8193603754043579, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.8502488136291504, "rewards/meter/std": 0.2549903690814972, "rewards/total_composite/mean": 0.8193603754043579, "rewards/total_composite/std": 0.250792533159256, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0295875072479248, "sampling/importance_sampling_ratio/min": 0.14527681469917297, "sampling/sampling_logp_difference/max": 1.9291143417358398, "sampling/sampling_logp_difference/mean": 0.17601756751537323, "step": 199 }, { "clip_ratio/high_max": 0.14806674886494875, "clip_ratio/high_mean": 0.14806674886494875, "clip_ratio/low_mean": 0.010593220591545105, "clip_ratio/low_min": 0.010593220591545105, "clip_ratio/region_mean": 0.15865996945649385, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 55.375, "completions/mean_terminated_length": 55.375, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 2.2072382867336273, "epoch": 0.007723200494284831, "frac_reward_zero_std": 0.0, "grad_norm": 14.159527778625488, "learning_rate": 9.396969696969697e-06, "loss": 0.0566, "num_tokens": 430070.0, "reward": 0.8980928659439087, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8980928659439087, "reward_meter_std": 0.20199771225452423, "reward_std": 0.20199769735336304, "reward_total_composite_mean": 0.8980928659439087, "reward_total_composite_std": 0.20199771225452423, "reward_total_mean": 0.8980928659439087, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8980928659439087, "rewards/meter/std": 0.20199771225452423, "rewards/total_composite/mean": 0.8980928659439087, "rewards/total_composite/std": 0.20199771225452423, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.031834602355957, "sampling/importance_sampling_ratio/min": 0.24137245118618011, "sampling/sampling_logp_difference/max": 1.4214141368865967, "sampling/sampling_logp_difference/mean": 0.17297667264938354, "step": 200 }, { "epoch": 0.007723200494284831, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.07692307692307693, "eval_completions/max_length": 457.3076923076923, "eval_completions/max_terminated_length": 368.0, "eval_completions/mean_length": 217.46153846153845, "eval_completions/mean_terminated_length": 192.05220383864182, "eval_completions/min_length": 56.0, "eval_completions/min_terminated_length": 56.0, "eval_entropy": 2.1048372708834133, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 430070.0, "eval_reward": 0.4151367247104645, "eval_reward_arabic_clean_mean": 0.9423076923076923, "eval_reward_arabic_clean_std": 0.12560976010102493, "eval_reward_count_adherence_mean": 0.9569021555093619, "eval_reward_count_adherence_std": 0.07193636994522351, "eval_reward_meter_mean": 0.4592640560406905, "eval_reward_meter_std": 0.3293467943484967, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.4151367247104645, "eval_reward_total_composite_std": 0.3096239452178662, "eval_reward_total_mean": 0.4151367247104645, "eval_rewards/arabic_clean/mean": 0.9423076923076923, "eval_rewards/arabic_clean/std": 0.12560976010102493, "eval_rewards/count_adherence/mean": 0.9569021555093619, "eval_rewards/count_adherence/std": 0.07193636994522351, "eval_rewards/meter/mean": 0.4592640560406905, "eval_rewards/meter/std": 0.3293467943484967, "eval_rewards/total_composite/mean": 0.4151367247104645, "eval_rewards/total_composite/std": 0.3096239452178662, "eval_runtime": 84.7428, "eval_samples_per_second": 1.227, "eval_sampling/importance_sampling_ratio/max": 1.5931972448642437, "eval_sampling/importance_sampling_ratio/mean": 1.0360724650896513, "eval_sampling/importance_sampling_ratio/min": 0.28642855469997114, "eval_sampling/sampling_logp_difference/max": 1.260967914874737, "eval_sampling/sampling_logp_difference/mean": 0.1303755704026956, "eval_steps_per_second": 0.153, "step": 200 }, { "clip_ratio/high_max": 0.08195106312632561, "clip_ratio/high_mean": 0.08195106312632561, "clip_ratio/low_mean": 0.0364175159484148, "clip_ratio/low_min": 0.0364175159484148, "clip_ratio/region_mean": 0.11836857907474041, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 31.875, "completions/mean_terminated_length": 31.875, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.9935045540332794, "epoch": 0.007761816496756255, "frac_reward_zero_std": 0.0, "grad_norm": 12.455371856689453, "learning_rate": 9.393939393939396e-06, "loss": 0.0034, "num_tokens": 431437.0, "reward": 0.9721503257751465, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9721503257751465, "reward_meter_std": 0.02047812007367611, "reward_std": 0.02047811821103096, "reward_total_composite_mean": 0.9721503257751465, "reward_total_composite_std": 0.02047812007367611, "reward_total_mean": 0.9721503257751465, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9721503257751465, "rewards/meter/std": 0.02047812007367611, "rewards/total_composite/mean": 0.9721503257751465, "rewards/total_composite/std": 0.02047812007367611, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007079839706421, "sampling/importance_sampling_ratio/min": 0.3022264838218689, "sampling/sampling_logp_difference/max": 1.1965785026550293, "sampling/sampling_logp_difference/mean": 0.13416853547096252, "step": 201 }, { "clip_ratio/high_max": 0.05716947093605995, "clip_ratio/high_mean": 0.05716947093605995, "clip_ratio/low_mean": 0.046742528676986694, "clip_ratio/low_min": 0.046742528676986694, "clip_ratio/region_mean": 0.10391199961304665, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 443.5, "completions/mean_terminated_length": 433.71429443359375, "completions/min_length": 371.0, "completions/min_terminated_length": 371.0, "entropy": 2.633182942867279, "epoch": 0.0078004324992276795, "frac_reward_zero_std": 0.0, "grad_norm": 2.5677430629730225, "learning_rate": 9.390909090909092e-06, "loss": -0.126, "num_tokens": 436473.0, "reward": 0.3806021213531494, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9318181872367859, "reward_count_adherence_std": 0.06428243219852448, "reward_meter_mean": 0.4077759385108948, "reward_meter_std": 0.2577785849571228, "reward_std": 0.23827768862247467, "reward_total_composite_mean": 0.3806021213531494, "reward_total_composite_std": 0.23827770352363586, "reward_total_mean": 0.3806021213531494, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9318181872367859, "rewards/count_adherence/std": 0.06428243219852448, "rewards/meter/mean": 0.4077759385108948, "rewards/meter/std": 0.2577785849571228, "rewards/total_composite/mean": 0.3806021213531494, "rewards/total_composite/std": 0.23827770352363586, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0397894382476807, "sampling/importance_sampling_ratio/min": 0.2143089473247528, "sampling/sampling_logp_difference/max": 1.5403366088867188, "sampling/sampling_logp_difference/mean": 0.1868906170129776, "step": 202 }, { "clip_ratio/high_max": 0.05425724573433399, "clip_ratio/high_mean": 0.05425724573433399, "clip_ratio/low_mean": 0.13778485730290413, "clip_ratio/low_min": 0.13778485730290413, "clip_ratio/region_mean": 0.19204210303723812, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 54.875, "completions/mean_terminated_length": 54.875, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 1.969245657324791, "epoch": 0.007839048501699104, "frac_reward_zero_std": 0.0, "grad_norm": 10.374756813049316, "learning_rate": 9.387878787878789e-06, "loss": 0.0063, "num_tokens": 438144.0, "reward": 0.18955467641353607, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.18955467641353607, "reward_meter_std": 0.2376922219991684, "reward_std": 0.2376922219991684, "reward_total_composite_mean": 0.18955467641353607, "reward_total_composite_std": 0.2376922219991684, "reward_total_mean": 0.18955467641353607, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.18955467641353607, "rewards/meter/std": 0.2376922219991684, "rewards/total_composite/mean": 0.18955467641353607, "rewards/total_composite/std": 0.2376922219991684, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0263407230377197, "sampling/importance_sampling_ratio/min": 0.24083620309829712, "sampling/sampling_logp_difference/max": 1.4236383438110352, "sampling/sampling_logp_difference/mean": 0.19148828089237213, "step": 203 }, { "clip_ratio/high_max": 0.05784035939723253, "clip_ratio/high_mean": 0.05784035939723253, "clip_ratio/low_mean": 0.05318944435566664, "clip_ratio/low_min": 0.05318944435566664, "clip_ratio/region_mean": 0.11102980375289917, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 316.75, "completions/mean_terminated_length": 316.75, "completions/min_length": 280.0, "completions/min_terminated_length": 280.0, "entropy": 1.8388848304748535, "epoch": 0.007877664504170528, "frac_reward_zero_std": 0.0, "grad_norm": 3.7568657398223877, "learning_rate": 9.384848484848486e-06, "loss": 0.0447, "num_tokens": 442326.0, "reward": 0.6847348213195801, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_meter_mean": 0.7115911245346069, "reward_meter_std": 0.30422648787498474, "reward_std": 0.28163081407546997, "reward_total_composite_mean": 0.6847348213195801, "reward_total_composite_std": 0.28163081407546997, "reward_total_mean": 0.6847348213195801, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/meter/mean": 0.7115911245346069, "rewards/meter/std": 0.30422648787498474, "rewards/total_composite/mean": 0.6847348213195801, "rewards/total_composite/std": 0.28163081407546997, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0315098762512207, "sampling/importance_sampling_ratio/min": 0.09864851087331772, "sampling/sampling_logp_difference/max": 2.316192150115967, "sampling/sampling_logp_difference/mean": 0.14612703025341034, "step": 204 }, { "clip_ratio/high_max": 0.04085497930645943, "clip_ratio/high_mean": 0.04085497930645943, "clip_ratio/low_mean": 0.04515550099313259, "clip_ratio/low_min": 0.04515550099313259, "clip_ratio/region_mean": 0.08601048029959202, "completions/clipped_ratio": 0.0, "completions/max_length": 22.0, "completions/max_terminated_length": 22.0, "completions/mean_length": 20.75, "completions/mean_terminated_length": 20.75, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.47789650596678257, "epoch": 0.007916280506641952, "frac_reward_zero_std": 0.0, "grad_norm": 22.554645538330078, "learning_rate": 9.381818181818183e-06, "loss": -0.0141, "num_tokens": 443612.0, "reward": 0.2569815218448639, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2569815218448639, "reward_meter_std": 0.17644450068473816, "reward_std": 0.17644448578357697, "reward_total_composite_mean": 0.2569815218448639, "reward_total_composite_std": 0.17644450068473816, "reward_total_mean": 0.2569815218448639, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2569815218448639, "rewards/meter/std": 0.17644450068473816, "rewards/total_composite/mean": 0.2569815218448639, "rewards/total_composite/std": 0.17644450068473816, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010235071182251, "sampling/importance_sampling_ratio/min": 0.09870558977127075, "sampling/sampling_logp_difference/max": 2.3156137466430664, "sampling/sampling_logp_difference/mean": 0.11848119646310806, "step": 205 }, { "clip_ratio/high_max": 0.08172544464468956, "clip_ratio/high_mean": 0.08172544464468956, "clip_ratio/low_mean": 0.02254154160618782, "clip_ratio/low_min": 0.02254154160618782, "clip_ratio/region_mean": 0.10426698625087738, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 502.0, "completions/mean_length": 459.125, "completions/mean_terminated_length": 451.5714416503906, "completions/min_length": 410.0, "completions/min_terminated_length": 410.0, "entropy": 2.3715617954730988, "epoch": 0.007954896509113376, "frac_reward_zero_std": 0.0, "grad_norm": 2.1156201362609863, "learning_rate": 9.378787878787879e-06, "loss": -0.2172, "num_tokens": 448589.0, "reward": 0.542489767074585, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8928571343421936, "reward_count_adherence_std": 0.05399491637945175, "reward_meter_mean": 0.5943694114685059, "reward_meter_std": 0.28580430150032043, "reward_std": 0.2698879539966583, "reward_total_composite_mean": 0.542489767074585, "reward_total_composite_std": 0.26988792419433594, "reward_total_mean": 0.542489767074585, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8928571343421936, "rewards/count_adherence/std": 0.05399491637945175, "rewards/meter/mean": 0.5943694114685059, "rewards/meter/std": 0.28580430150032043, "rewards/total_composite/mean": 0.542489767074585, "rewards/total_composite/std": 0.26988792419433594, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0428776741027832, "sampling/importance_sampling_ratio/min": 0.2488904893398285, "sampling/sampling_logp_difference/max": 1.390742301940918, "sampling/sampling_logp_difference/mean": 0.16989928483963013, "step": 206 }, { "clip_ratio/high_max": 0.09698447678238153, "clip_ratio/high_mean": 0.09698447678238153, "clip_ratio/low_mean": 0.03460816852748394, "clip_ratio/low_min": 0.03460816852748394, "clip_ratio/region_mean": 0.13159264530986547, "completions/clipped_ratio": 0.0, "completions/max_length": 223.0, "completions/max_terminated_length": 223.0, "completions/mean_length": 200.625, "completions/mean_terminated_length": 200.625, "completions/min_length": 185.0, "completions/min_terminated_length": 185.0, "entropy": 2.5693641006946564, "epoch": 0.0079935125115848, "frac_reward_zero_std": 0.0, "grad_norm": 5.197041988372803, "learning_rate": 9.375757575757576e-06, "loss": 0.0262, "num_tokens": 451666.0, "reward": 0.5297455787658691, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.5436156988143921, "reward_meter_std": 0.36589959263801575, "reward_std": 0.3861840069293976, "reward_total_composite_mean": 0.5297455787658691, "reward_total_composite_std": 0.3861840069293976, "reward_total_mean": 0.5297455787658691, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.5436156988143921, "rewards/meter/std": 0.36589959263801575, "rewards/total_composite/mean": 0.5297455787658691, "rewards/total_composite/std": 0.3861840069293976, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0335437059402466, "sampling/importance_sampling_ratio/min": 0.25058552622795105, "sampling/sampling_logp_difference/max": 1.3839550018310547, "sampling/sampling_logp_difference/mean": 0.17878292500972748, "step": 207 }, { "clip_ratio/high_max": 0.08172481320798397, "clip_ratio/high_mean": 0.08172481320798397, "clip_ratio/low_mean": 0.04633831046521664, "clip_ratio/low_min": 0.04633831046521664, "clip_ratio/region_mean": 0.1280631236732006, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 112.0, "completions/mean_terminated_length": 54.857147216796875, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 1.829095020890236, "epoch": 0.008032128514056224, "frac_reward_zero_std": 0.0, "grad_norm": 4.907017707824707, "learning_rate": 9.372727272727273e-06, "loss": -0.0625, "num_tokens": 453242.0, "reward": 0.5157449841499329, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.5209001302719116, "reward_meter_std": 0.4826778173446655, "reward_std": 0.48821765184402466, "reward_total_composite_mean": 0.5157449841499329, "reward_total_composite_std": 0.48821765184402466, "reward_total_mean": 0.5157449841499329, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.5209001302719116, "rewards/meter/std": 0.4826778173446655, "rewards/total_composite/mean": 0.5157449841499329, "rewards/total_composite/std": 0.48821765184402466, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0225262641906738, "sampling/importance_sampling_ratio/min": 0.16763810813426971, "sampling/sampling_logp_difference/max": 1.7859477996826172, "sampling/sampling_logp_difference/mean": 0.1978602260351181, "step": 208 }, { "clip_ratio/high_max": 0.07205317448824644, "clip_ratio/high_mean": 0.07205317448824644, "clip_ratio/low_mean": 0.05209819972515106, "clip_ratio/low_min": 0.05209819972515106, "clip_ratio/region_mean": 0.1241513742133975, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 65.125, "completions/mean_terminated_length": 65.125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 1.5745535343885422, "epoch": 0.008070744516527648, "frac_reward_zero_std": 0.0, "grad_norm": 10.775023460388184, "learning_rate": 9.36969696969697e-06, "loss": 0.0425, "num_tokens": 455075.0, "reward": 0.687164843082428, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.687164843082428, "reward_meter_std": 0.42298591136932373, "reward_std": 0.42298588156700134, "reward_total_composite_mean": 0.687164843082428, "reward_total_composite_std": 0.42298591136932373, "reward_total_mean": 0.687164843082428, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.687164843082428, "rewards/meter/std": 0.42298591136932373, "rewards/total_composite/mean": 0.687164843082428, "rewards/total_composite/std": 0.42298591136932373, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.031755805015564, "sampling/importance_sampling_ratio/min": 0.2617337703704834, "sampling/sampling_logp_difference/max": 1.3404273986816406, "sampling/sampling_logp_difference/mean": 0.1489512026309967, "step": 209 }, { "clip_ratio/high_max": 0.0914016654714942, "clip_ratio/high_mean": 0.0914016654714942, "clip_ratio/low_mean": 0.06012810207903385, "clip_ratio/low_min": 0.06012810207903385, "clip_ratio/region_mean": 0.15152976755052805, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 94.75, "completions/mean_terminated_length": 94.75, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 2.1869092285633087, "epoch": 0.008109360518999072, "frac_reward_zero_std": 0.0, "grad_norm": 8.813285827636719, "learning_rate": 9.366666666666668e-06, "loss": 0.0918, "num_tokens": 457201.0, "reward": 0.8014340400695801, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8014340400695801, "reward_meter_std": 0.26059165596961975, "reward_std": 0.26059165596961975, "reward_total_composite_mean": 0.8014340400695801, "reward_total_composite_std": 0.26059165596961975, "reward_total_mean": 0.8014340400695801, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8014340400695801, "rewards/meter/std": 0.26059165596961975, "rewards/total_composite/mean": 0.8014340400695801, "rewards/total_composite/std": 0.26059165596961975, "sampling/importance_sampling_ratio/max": 1.846545934677124, "sampling/importance_sampling_ratio/mean": 1.0492236614227295, "sampling/importance_sampling_ratio/min": 0.22778037190437317, "sampling/sampling_logp_difference/max": 1.4793734550476074, "sampling/sampling_logp_difference/mean": 0.16893059015274048, "step": 210 }, { "clip_ratio/high_max": 0.06869588885456324, "clip_ratio/high_mean": 0.06869588885456324, "clip_ratio/low_mean": 0.0616428735665977, "clip_ratio/low_min": 0.0616428735665977, "clip_ratio/region_mean": 0.13033876242116094, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 31.0, "completions/mean_terminated_length": 31.0, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 2.1823814064264297, "epoch": 0.008147976521470498, "frac_reward_zero_std": 0.0, "grad_norm": 14.590749740600586, "learning_rate": 9.363636363636365e-06, "loss": 0.043, "num_tokens": 458609.0, "reward": 0.3326827883720398, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3326827883720398, "reward_meter_std": 0.391814649105072, "reward_std": 0.39181461930274963, "reward_total_composite_mean": 0.3326827883720398, "reward_total_composite_std": 0.391814649105072, "reward_total_mean": 0.3326827883720398, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3326827883720398, "rewards/meter/std": 0.391814649105072, "rewards/total_composite/mean": 0.3326827883720398, "rewards/total_composite/std": 0.391814649105072, "sampling/importance_sampling_ratio/max": 1.9337586164474487, "sampling/importance_sampling_ratio/mean": 1.0280555486679077, "sampling/importance_sampling_ratio/min": 0.3309429883956909, "sampling/sampling_logp_difference/max": 1.105809211730957, "sampling/sampling_logp_difference/mean": 0.17955826222896576, "step": 211 }, { "clip_ratio/high_max": 0.0721238311380148, "clip_ratio/high_mean": 0.0721238311380148, "clip_ratio/low_mean": 0.10219099884852767, "clip_ratio/low_min": 0.10219099884852767, "clip_ratio/region_mean": 0.17431482998654246, "completions/clipped_ratio": 0.0, "completions/max_length": 51.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 35.125, "completions/mean_terminated_length": 35.125, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 1.3654158115386963, "epoch": 0.008186592523941922, "frac_reward_zero_std": 0.0, "grad_norm": 20.575647354125977, "learning_rate": 9.36060606060606e-06, "loss": 0.1396, "num_tokens": 460114.0, "reward": 0.45719897747039795, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.45719897747039795, "reward_meter_std": 0.3513868451118469, "reward_std": 0.35138681530952454, "reward_total_composite_mean": 0.45719897747039795, "reward_total_composite_std": 0.3513868451118469, "reward_total_mean": 0.45719897747039795, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.45719897747039795, "rewards/meter/std": 0.3513868451118469, "rewards/total_composite/mean": 0.45719897747039795, "rewards/total_composite/std": 0.3513868451118469, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9953309297561646, "sampling/importance_sampling_ratio/min": 0.1695476919412613, "sampling/sampling_logp_difference/max": 1.7746210098266602, "sampling/sampling_logp_difference/mean": 0.18254390358924866, "step": 212 }, { "clip_ratio/high_max": 0.09881766140460968, "clip_ratio/high_mean": 0.09881766140460968, "clip_ratio/low_mean": 0.039724151603877544, "clip_ratio/low_min": 0.039724151603877544, "clip_ratio/region_mean": 0.13854181300848722, "completions/clipped_ratio": 0.0, "completions/max_length": 119.0, "completions/max_terminated_length": 119.0, "completions/mean_length": 106.25, "completions/mean_terminated_length": 106.25, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 2.2049818336963654, "epoch": 0.008225208526413346, "frac_reward_zero_std": 0.0, "grad_norm": 7.170069694519043, "learning_rate": 9.357575757575758e-06, "loss": 0.0285, "num_tokens": 462364.0, "reward": 0.6657600998878479, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6657600998878479, "reward_meter_std": 0.4287368059158325, "reward_std": 0.4287368059158325, "reward_total_composite_mean": 0.6657600998878479, "reward_total_composite_std": 0.4287368059158325, "reward_total_mean": 0.6657600998878479, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6657600998878479, "rewards/meter/std": 0.4287368059158325, "rewards/total_composite/mean": 0.6657600998878479, "rewards/total_composite/std": 0.4287368059158325, "sampling/importance_sampling_ratio/max": 1.96308434009552, "sampling/importance_sampling_ratio/mean": 1.0314472913742065, "sampling/importance_sampling_ratio/min": 0.20997050404548645, "sampling/sampling_logp_difference/max": 1.5607881546020508, "sampling/sampling_logp_difference/mean": 0.16524526476860046, "step": 213 }, { "clip_ratio/high_max": 0.08606619853526354, "clip_ratio/high_mean": 0.08606619853526354, "clip_ratio/low_mean": 0.02658250369131565, "clip_ratio/low_min": 0.02658250369131565, "clip_ratio/region_mean": 0.11264870222657919, "completions/clipped_ratio": 0.0, "completions/max_length": 44.0, "completions/max_terminated_length": 44.0, "completions/mean_length": 37.25, "completions/mean_terminated_length": 37.25, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 1.585478514432907, "epoch": 0.00826382452888477, "frac_reward_zero_std": 0.0, "grad_norm": 12.966662406921387, "learning_rate": 9.354545454545455e-06, "loss": 0.0144, "num_tokens": 463910.0, "reward": 0.9338778257369995, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9338778257369995, "reward_meter_std": 0.12218131124973297, "reward_std": 0.12218130379915237, "reward_total_composite_mean": 0.9338778257369995, "reward_total_composite_std": 0.12218131124973297, "reward_total_mean": 0.9338778257369995, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9338778257369995, "rewards/meter/std": 0.12218131124973297, "rewards/total_composite/mean": 0.9338778257369995, "rewards/total_composite/std": 0.12218131124973297, "sampling/importance_sampling_ratio/max": 1.8531534671783447, "sampling/importance_sampling_ratio/mean": 1.0329978466033936, "sampling/importance_sampling_ratio/min": 0.2841220498085022, "sampling/sampling_logp_difference/max": 1.2583513259887695, "sampling/sampling_logp_difference/mean": 0.15117491781711578, "step": 214 }, { "clip_ratio/high_max": 0.09631683025509119, "clip_ratio/high_mean": 0.09631683025509119, "clip_ratio/low_mean": 0.04079106356948614, "clip_ratio/low_min": 0.04079106356948614, "clip_ratio/region_mean": 0.13710789382457733, "completions/clipped_ratio": 0.0, "completions/max_length": 46.0, "completions/max_terminated_length": 46.0, "completions/mean_length": 34.375, "completions/mean_terminated_length": 34.375, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 1.831810086965561, "epoch": 0.008302440531356195, "frac_reward_zero_std": 0.0, "grad_norm": 16.243061065673828, "learning_rate": 9.351515151515152e-06, "loss": 0.239, "num_tokens": 465457.0, "reward": 0.6347050666809082, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.6347050666809082, "reward_meter_std": 0.485779345035553, "reward_std": 0.485779345035553, "reward_total_composite_mean": 0.6347050666809082, "reward_total_composite_std": 0.485779345035553, "reward_total_mean": 0.6347050666809082, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.6347050666809082, "rewards/meter/std": 0.485779345035553, "rewards/total_composite/mean": 0.6347050666809082, "rewards/total_composite/std": 0.485779345035553, "sampling/importance_sampling_ratio/max": 1.7199817895889282, "sampling/importance_sampling_ratio/mean": 1.0188474655151367, "sampling/importance_sampling_ratio/min": 0.31832820177078247, "sampling/sampling_logp_difference/max": 1.1446723937988281, "sampling/sampling_logp_difference/mean": 0.17325210571289062, "step": 215 }, { "clip_ratio/high_max": 0.015360169112682343, "clip_ratio/high_mean": 0.015360169112682343, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.015360169112682343, "completions/clipped_ratio": 0.875, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 507.0, "completions/mean_terminated_length": 472.0, "completions/min_length": 472.0, "completions/min_terminated_length": 472.0, "entropy": 0.36599498987197876, "epoch": 0.008341056533827619, "frac_reward_zero_std": 0.0, "grad_norm": 0.7487542033195496, "learning_rate": 9.34848484848485e-06, "loss": -0.0905, "num_tokens": 467753.0, "reward": 0.18635782599449158, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.637499988079071, "reward_count_adherence_std": 0.06408698856830597, "reward_meter_mean": 0.3023744523525238, "reward_meter_std": 0.17562586069107056, "reward_std": 0.1041322648525238, "reward_total_composite_mean": 0.18635782599449158, "reward_total_composite_std": 0.1041322648525238, "reward_total_mean": 0.18635782599449158, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.637499988079071, "rewards/count_adherence/std": 0.06408698856830597, "rewards/meter/mean": 0.3023744523525238, "rewards/meter/std": 0.17562586069107056, "rewards/total_composite/mean": 0.18635782599449158, "rewards/total_composite/std": 0.1041322648525238, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.041240930557251, "sampling/importance_sampling_ratio/min": 0.2798839509487152, "sampling/sampling_logp_difference/max": 1.2733802795410156, "sampling/sampling_logp_difference/mean": 0.18534618616104126, "step": 216 }, { "clip_ratio/high_max": 0.0961015336215496, "clip_ratio/high_mean": 0.0961015336215496, "clip_ratio/low_mean": 0.08761653117835522, "clip_ratio/low_min": 0.08761653117835522, "clip_ratio/region_mean": 0.18371806479990482, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 52.0, "completions/mean_terminated_length": 52.0, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 2.0648878514766693, "epoch": 0.008379672536299043, "frac_reward_zero_std": 0.0, "grad_norm": 11.847907066345215, "learning_rate": 9.345454545454547e-06, "loss": -0.0148, "num_tokens": 469457.0, "reward": 0.5340137481689453, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5340137481689453, "reward_meter_std": 0.3664604425430298, "reward_std": 0.3664604127407074, "reward_total_composite_mean": 0.5340137481689453, "reward_total_composite_std": 0.3664604425430298, "reward_total_mean": 0.5340137481689453, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5340137481689453, "rewards/meter/std": 0.3664604425430298, "rewards/total_composite/mean": 0.5340137481689453, "rewards/total_composite/std": 0.3664604425430298, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0360443592071533, "sampling/importance_sampling_ratio/min": 0.1973160207271576, "sampling/sampling_logp_difference/max": 1.6229486465454102, "sampling/sampling_logp_difference/mean": 0.19996923208236694, "step": 217 }, { "clip_ratio/high_max": 0.07770076114684343, "clip_ratio/high_mean": 0.07770076114684343, "clip_ratio/low_mean": 0.05961098615080118, "clip_ratio/low_min": 0.05961098615080118, "clip_ratio/region_mean": 0.13731174729764462, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 105.5, "completions/mean_terminated_length": 105.5, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 1.9123822152614594, "epoch": 0.008418288538770467, "frac_reward_zero_std": 0.0, "grad_norm": 8.575998306274414, "learning_rate": 9.342424242424243e-06, "loss": -0.0238, "num_tokens": 471653.0, "reward": 0.5959634184837341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5959634184837341, "reward_meter_std": 0.32533523440361023, "reward_std": 0.32533523440361023, "reward_total_composite_mean": 0.5959634184837341, "reward_total_composite_std": 0.32533523440361023, "reward_total_mean": 0.5959634184837341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5959634184837341, "rewards/meter/std": 0.32533523440361023, "rewards/total_composite/mean": 0.5959634184837341, "rewards/total_composite/std": 0.32533523440361023, "sampling/importance_sampling_ratio/max": 1.9590723514556885, "sampling/importance_sampling_ratio/mean": 1.0328844785690308, "sampling/importance_sampling_ratio/min": 0.2482338696718216, "sampling/sampling_logp_difference/max": 1.3933839797973633, "sampling/sampling_logp_difference/mean": 0.16784748435020447, "step": 218 }, { "clip_ratio/high_max": 0.08742227591574192, "clip_ratio/high_mean": 0.08742227591574192, "clip_ratio/low_mean": 0.07042216509580612, "clip_ratio/low_min": 0.07042216509580612, "clip_ratio/region_mean": 0.15784444101154804, "completions/clipped_ratio": 0.0, "completions/max_length": 136.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 132.125, "completions/mean_terminated_length": 132.125, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 1.5948337465524673, "epoch": 0.008456904541241891, "frac_reward_zero_std": 0.0, "grad_norm": 7.932456016540527, "learning_rate": 9.33939393939394e-06, "loss": 0.0253, "num_tokens": 474198.0, "reward": 0.4994851350784302, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4994851350784302, "reward_meter_std": 0.3480455279350281, "reward_std": 0.3480455279350281, "reward_total_composite_mean": 0.4994851350784302, "reward_total_composite_std": 0.3480455279350281, "reward_total_mean": 0.4994851350784302, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4994851350784302, "rewards/meter/std": 0.3480455279350281, "rewards/total_composite/mean": 0.4994851350784302, "rewards/total_composite/std": 0.3480455279350281, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0123792886734009, "sampling/importance_sampling_ratio/min": 0.17448177933692932, "sampling/sampling_logp_difference/max": 1.7459349632263184, "sampling/sampling_logp_difference/mean": 0.17217673361301422, "step": 219 }, { "clip_ratio/high_max": 0.06848039291799068, "clip_ratio/high_mean": 0.06848039291799068, "clip_ratio/low_mean": 0.06260535214096308, "clip_ratio/low_min": 0.06260535214096308, "clip_ratio/region_mean": 0.13108574505895376, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 1.6451235264539719, "epoch": 0.008495520543713315, "frac_reward_zero_std": 0.0, "grad_norm": 9.201763153076172, "learning_rate": 9.336363636363637e-06, "loss": -0.0252, "num_tokens": 475969.0, "reward": 0.5444612503051758, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5444612503051758, "reward_meter_std": 0.46134647727012634, "reward_std": 0.46134647727012634, "reward_total_composite_mean": 0.5444612503051758, "reward_total_composite_std": 0.46134647727012634, "reward_total_mean": 0.5444612503051758, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5444612503051758, "rewards/meter/std": 0.46134647727012634, "rewards/total_composite/mean": 0.5444612503051758, "rewards/total_composite/std": 0.46134647727012634, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0116194486618042, "sampling/importance_sampling_ratio/min": 0.31130722165107727, "sampling/sampling_logp_difference/max": 1.1669750213623047, "sampling/sampling_logp_difference/mean": 0.15713398158550262, "step": 220 }, { "clip_ratio/high_max": 0.07579075917601585, "clip_ratio/high_mean": 0.07579075917601585, "clip_ratio/low_mean": 0.07524601556360722, "clip_ratio/low_min": 0.07524601556360722, "clip_ratio/region_mean": 0.15103677473962307, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 58.625, "completions/mean_terminated_length": 58.625, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 1.5837604850530624, "epoch": 0.00853413654618474, "frac_reward_zero_std": 0.0, "grad_norm": 11.238251686096191, "learning_rate": 9.333333333333334e-06, "loss": -0.0093, "num_tokens": 477742.0, "reward": 0.5600961446762085, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5600961446762085, "reward_meter_std": 0.4695318341255188, "reward_std": 0.4695318341255188, "reward_total_composite_mean": 0.5600961446762085, "reward_total_composite_std": 0.4695318341255188, "reward_total_mean": 0.5600961446762085, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5600961446762085, "rewards/meter/std": 0.4695318341255188, "rewards/total_composite/mean": 0.5600961446762085, "rewards/total_composite/std": 0.4695318341255188, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.021318793296814, "sampling/importance_sampling_ratio/min": 0.14270596206188202, "sampling/sampling_logp_difference/max": 1.9469690322875977, "sampling/sampling_logp_difference/mean": 0.16859595477581024, "step": 221 }, { "clip_ratio/high_max": 0.09998656460084021, "clip_ratio/high_mean": 0.09998656460084021, "clip_ratio/low_mean": 0.022727273404598236, "clip_ratio/low_min": 0.022727273404598236, "clip_ratio/region_mean": 0.12271383800543845, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 36.0, "completions/mean_terminated_length": 36.0, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 1.5899460166692734, "epoch": 0.008572752548656163, "frac_reward_zero_std": 0.0, "grad_norm": 11.750017166137695, "learning_rate": 9.33030303030303e-06, "loss": -0.147, "num_tokens": 479254.0, "reward": 0.8671466112136841, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8671466112136841, "reward_meter_std": 0.3500947654247284, "reward_std": 0.3500947654247284, "reward_total_composite_mean": 0.8671466112136841, "reward_total_composite_std": 0.3500947654247284, "reward_total_mean": 0.8671466112136841, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8671466112136841, "rewards/meter/std": 0.3500947654247284, "rewards/total_composite/mean": 0.8671466112136841, "rewards/total_composite/std": 0.3500947654247284, "sampling/importance_sampling_ratio/max": 1.9629331827163696, "sampling/importance_sampling_ratio/mean": 1.0165225267410278, "sampling/importance_sampling_ratio/min": 0.13546092808246613, "sampling/sampling_logp_difference/max": 1.9990720748901367, "sampling/sampling_logp_difference/mean": 0.16731053590774536, "step": 222 }, { "clip_ratio/high_max": 0.07255261763930321, "clip_ratio/high_mean": 0.07255261763930321, "clip_ratio/low_mean": 0.05262349545955658, "clip_ratio/low_min": 0.05262349545955658, "clip_ratio/region_mean": 0.1251761130988598, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 73.75, "completions/mean_terminated_length": 73.75, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 1.9959132969379425, "epoch": 0.008611368551127587, "frac_reward_zero_std": 0.0, "grad_norm": 7.995184421539307, "learning_rate": 9.327272727272729e-06, "loss": -0.0678, "num_tokens": 481092.0, "reward": 0.8009564280509949, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8009564280509949, "reward_meter_std": 0.26855042576789856, "reward_std": 0.26855039596557617, "reward_total_composite_mean": 0.8009564280509949, "reward_total_composite_std": 0.26855042576789856, "reward_total_mean": 0.8009564280509949, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8009564280509949, "rewards/meter/std": 0.26855042576789856, "rewards/total_composite/mean": 0.8009564280509949, "rewards/total_composite/std": 0.26855042576789856, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0155538320541382, "sampling/importance_sampling_ratio/min": 0.2769174873828888, "sampling/sampling_logp_difference/max": 1.2840356826782227, "sampling/sampling_logp_difference/mean": 0.17098720371723175, "step": 223 }, { "clip_ratio/high_max": 0.10716481134295464, "clip_ratio/high_mean": 0.10716481134295464, "clip_ratio/low_mean": 0.04744089301675558, "clip_ratio/low_min": 0.04744089301675558, "clip_ratio/region_mean": 0.15460570435971022, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 102.125, "completions/mean_terminated_length": 102.125, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 1.7342596799135208, "epoch": 0.008649984553599012, "frac_reward_zero_std": 0.0, "grad_norm": 8.910178184509277, "learning_rate": 9.324242424242424e-06, "loss": 0.0128, "num_tokens": 483221.0, "reward": 0.7856420278549194, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7856420278549194, "reward_meter_std": 0.2833492159843445, "reward_std": 0.2833492159843445, "reward_total_composite_mean": 0.7856420278549194, "reward_total_composite_std": 0.2833492159843445, "reward_total_mean": 0.7856420278549194, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7856420278549194, "rewards/meter/std": 0.2833492159843445, "rewards/total_composite/mean": 0.7856420278549194, "rewards/total_composite/std": 0.2833492159843445, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.028478741645813, "sampling/importance_sampling_ratio/min": 0.1979585438966751, "sampling/sampling_logp_difference/max": 1.6196975708007812, "sampling/sampling_logp_difference/mean": 0.16320450603961945, "step": 224 }, { "clip_ratio/high_max": 0.0867786854505539, "clip_ratio/high_mean": 0.0867786854505539, "clip_ratio/low_mean": 0.04819977842271328, "clip_ratio/low_min": 0.04819977842271328, "clip_ratio/region_mean": 0.13497846387326717, "completions/clipped_ratio": 0.0, "completions/max_length": 211.0, "completions/max_terminated_length": 211.0, "completions/mean_length": 190.375, "completions/mean_terminated_length": 190.375, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "entropy": 2.218223586678505, "epoch": 0.008688600556070436, "frac_reward_zero_std": 0.0, "grad_norm": 5.88389253616333, "learning_rate": 9.321212121212122e-06, "loss": 0.0627, "num_tokens": 486360.0, "reward": 0.6395675539970398, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_meter_mean": 0.680176854133606, "reward_meter_std": 0.38275107741355896, "reward_std": 0.3551376461982727, "reward_total_composite_mean": 0.6395675539970398, "reward_total_composite_std": 0.3551376760005951, "reward_total_mean": 0.6395675539970398, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/meter/mean": 0.680176854133606, "rewards/meter/std": 0.38275107741355896, "rewards/total_composite/mean": 0.6395675539970398, "rewards/total_composite/std": 0.3551376760005951, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.033612608909607, "sampling/importance_sampling_ratio/min": 0.1116233542561531, "sampling/sampling_logp_difference/max": 2.192625045776367, "sampling/sampling_logp_difference/mean": 0.16371257603168488, "step": 225 }, { "clip_ratio/high_max": 0.09920443315058947, "clip_ratio/high_mean": 0.09920443315058947, "clip_ratio/low_mean": 0.030824373476207256, "clip_ratio/low_min": 0.030824373476207256, "clip_ratio/region_mean": 0.13002880662679672, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 157.75, "completions/mean_terminated_length": 157.75, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 1.9846401810646057, "epoch": 0.00872721655854186, "frac_reward_zero_std": 0.0, "grad_norm": 4.9813737869262695, "learning_rate": 9.318181818181819e-06, "loss": 0.0006, "num_tokens": 489110.0, "reward": 0.8852719068527222, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8852719068527222, "reward_meter_std": 0.17902569472789764, "reward_std": 0.17902567982673645, "reward_total_composite_mean": 0.8852719068527222, "reward_total_composite_std": 0.17902569472789764, "reward_total_mean": 0.8852719068527222, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8852719068527222, "rewards/meter/std": 0.17902569472789764, "rewards/total_composite/mean": 0.8852719068527222, "rewards/total_composite/std": 0.17902569472789764, "sampling/importance_sampling_ratio/max": 1.984347939491272, "sampling/importance_sampling_ratio/mean": 1.0297491550445557, "sampling/importance_sampling_ratio/min": 0.2885850667953491, "sampling/sampling_logp_difference/max": 1.2427654266357422, "sampling/sampling_logp_difference/mean": 0.14676453173160553, "step": 226 }, { "clip_ratio/high_max": 0.09229664131999016, "clip_ratio/high_mean": 0.09229664131999016, "clip_ratio/low_mean": 0.07808315567672253, "clip_ratio/low_min": 0.07808315567672253, "clip_ratio/region_mean": 0.17037979699671268, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 56.25, "completions/mean_terminated_length": 56.25, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 2.041958436369896, "epoch": 0.008765832561013284, "frac_reward_zero_std": 0.0, "grad_norm": 10.270444869995117, "learning_rate": 9.315151515151516e-06, "loss": 0.1164, "num_tokens": 490872.0, "reward": 0.36732152104377747, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.36732152104377747, "reward_meter_std": 0.30656546354293823, "reward_std": 0.30656546354293823, "reward_total_composite_mean": 0.36732152104377747, "reward_total_composite_std": 0.30656546354293823, "reward_total_mean": 0.36732152104377747, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.36732152104377747, "rewards/meter/std": 0.30656546354293823, "rewards/total_composite/mean": 0.36732152104377747, "rewards/total_composite/std": 0.30656546354293823, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0084258317947388, "sampling/importance_sampling_ratio/min": 0.21239109337329865, "sampling/sampling_logp_difference/max": 1.549325942993164, "sampling/sampling_logp_difference/mean": 0.18055380880832672, "step": 227 }, { "clip_ratio/high_max": 0.1098766503855586, "clip_ratio/high_mean": 0.1098766503855586, "clip_ratio/low_mean": 0.027772237546741962, "clip_ratio/low_min": 0.027772237546741962, "clip_ratio/region_mean": 0.13764888793230057, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 91.25, "completions/mean_terminated_length": 91.25, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 1.5294092446565628, "epoch": 0.008804448563484708, "frac_reward_zero_std": 0.0, "grad_norm": 9.097184181213379, "learning_rate": 9.312121212121212e-06, "loss": 0.0772, "num_tokens": 492930.0, "reward": 0.8796967267990112, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8796967267990112, "reward_meter_std": 0.2009752243757248, "reward_std": 0.2009752094745636, "reward_total_composite_mean": 0.8796967267990112, "reward_total_composite_std": 0.2009752243757248, "reward_total_mean": 0.8796967267990112, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8796967267990112, "rewards/meter/std": 0.2009752243757248, "rewards/total_composite/mean": 0.8796967267990112, "rewards/total_composite/std": 0.2009752243757248, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.02587890625, "sampling/importance_sampling_ratio/min": 0.15230019390583038, "sampling/sampling_logp_difference/max": 1.881901741027832, "sampling/sampling_logp_difference/mean": 0.14313152432441711, "step": 228 }, { "clip_ratio/high_max": 0.04163628350943327, "clip_ratio/high_mean": 0.04163628350943327, "clip_ratio/low_mean": 0.0690256292000413, "clip_ratio/low_min": 0.0690256292000413, "clip_ratio/region_mean": 0.11066191270947456, "completions/clipped_ratio": 0.0, "completions/max_length": 232.0, "completions/max_terminated_length": 232.0, "completions/mean_length": 215.0, "completions/mean_terminated_length": 215.0, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 1.8009164333343506, "epoch": 0.008843064565956132, "frac_reward_zero_std": 0.0, "grad_norm": 4.891687393188477, "learning_rate": 9.30909090909091e-06, "loss": 0.0304, "num_tokens": 496362.0, "reward": 0.4802461862564087, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.5281928777694702, "reward_meter_std": 0.36326658725738525, "reward_std": 0.3348220884799957, "reward_total_composite_mean": 0.4802461862564087, "reward_total_composite_std": 0.3348220884799957, "reward_total_mean": 0.4802461862564087, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.5281928777694702, "rewards/meter/std": 0.36326658725738525, "rewards/total_composite/mean": 0.4802461862564087, "rewards/total_composite/std": 0.3348220884799957, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.028792381286621, "sampling/importance_sampling_ratio/min": 0.24842379987239838, "sampling/sampling_logp_difference/max": 1.3926191329956055, "sampling/sampling_logp_difference/mean": 0.14938952028751373, "step": 229 }, { "clip_ratio/high_max": 0.12172973342239857, "clip_ratio/high_mean": 0.12172973342239857, "clip_ratio/low_mean": 0.03780912980437279, "clip_ratio/low_min": 0.03780912980437279, "clip_ratio/region_mean": 0.15953886322677135, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 117.625, "completions/mean_terminated_length": 117.625, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 1.9786228984594345, "epoch": 0.008881680568427556, "frac_reward_zero_std": 0.0, "grad_norm": 8.536412239074707, "learning_rate": 9.306060606060608e-06, "loss": -0.0263, "num_tokens": 498671.0, "reward": 0.723436713218689, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.723436713218689, "reward_meter_std": 0.3901357650756836, "reward_std": 0.3901357650756836, "reward_total_composite_mean": 0.723436713218689, "reward_total_composite_std": 0.3901357650756836, "reward_total_mean": 0.723436713218689, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.723436713218689, "rewards/meter/std": 0.3901357650756836, "rewards/total_composite/mean": 0.723436713218689, "rewards/total_composite/std": 0.3901357650756836, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0375686883926392, "sampling/importance_sampling_ratio/min": 0.3272934556007385, "sampling/sampling_logp_difference/max": 1.1610007286071777, "sampling/sampling_logp_difference/mean": 0.17858588695526123, "step": 230 }, { "clip_ratio/high_max": 0.07807724550366402, "clip_ratio/high_mean": 0.07807724550366402, "clip_ratio/low_mean": 0.06935460586100817, "clip_ratio/low_min": 0.06935460586100817, "clip_ratio/region_mean": 0.14743185136467218, "completions/clipped_ratio": 0.0, "completions/max_length": 126.0, "completions/max_terminated_length": 126.0, "completions/mean_length": 115.75, "completions/mean_terminated_length": 115.75, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 2.1323368698358536, "epoch": 0.00892029657089898, "frac_reward_zero_std": 0.0, "grad_norm": 6.538661003112793, "learning_rate": 9.303030303030303e-06, "loss": 0.0366, "num_tokens": 500877.0, "reward": 0.6978538036346436, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6978538036346436, "reward_meter_std": 0.3173922300338745, "reward_std": 0.3173922002315521, "reward_total_composite_mean": 0.6978538036346436, "reward_total_composite_std": 0.3173922300338745, "reward_total_mean": 0.6978538036346436, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6978538036346436, "rewards/meter/std": 0.3173922300338745, "rewards/total_composite/mean": 0.6978538036346436, "rewards/total_composite/std": 0.3173922300338745, "sampling/importance_sampling_ratio/max": 1.9082386493682861, "sampling/importance_sampling_ratio/mean": 1.036413550376892, "sampling/importance_sampling_ratio/min": 0.2662426829338074, "sampling/sampling_logp_difference/max": 1.3233470916748047, "sampling/sampling_logp_difference/mean": 0.15994611382484436, "step": 231 }, { "clip_ratio/high_max": 0.12983744032680988, "clip_ratio/high_mean": 0.12983744032680988, "clip_ratio/low_mean": 0.03656993992626667, "clip_ratio/low_min": 0.03656993992626667, "clip_ratio/region_mean": 0.16640738025307655, "completions/clipped_ratio": 0.0, "completions/max_length": 168.0, "completions/max_terminated_length": 168.0, "completions/mean_length": 147.875, "completions/mean_terminated_length": 147.875, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 1.93074631690979, "epoch": 0.008958912573370404, "frac_reward_zero_std": 0.0, "grad_norm": 6.832438945770264, "learning_rate": 9.3e-06, "loss": 0.0737, "num_tokens": 503508.0, "reward": 0.7425558567047119, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.7667824029922485, "reward_meter_std": 0.29820093512535095, "reward_std": 0.2870854139328003, "reward_total_composite_mean": 0.7425558567047119, "reward_total_composite_std": 0.2870854437351227, "reward_total_mean": 0.7425558567047119, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.7667824029922485, "rewards/meter/std": 0.29820093512535095, "rewards/total_composite/mean": 0.7425558567047119, "rewards/total_composite/std": 0.2870854437351227, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.031022071838379, "sampling/importance_sampling_ratio/min": 0.20527270436286926, "sampling/sampling_logp_difference/max": 1.5834159851074219, "sampling/sampling_logp_difference/mean": 0.1671251356601715, "step": 232 }, { "clip_ratio/high_max": 0.023773369379341602, "clip_ratio/high_mean": 0.023773369379341602, "clip_ratio/low_mean": 0.009834368713200092, "clip_ratio/low_min": 0.009834368713200092, "clip_ratio/region_mean": 0.033607738092541695, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 502.5, "completions/mean_terminated_length": 486.66668701171875, "completions/min_length": 480.0, "completions/min_terminated_length": 480.0, "entropy": 0.6105214804410934, "epoch": 0.008997528575841829, "frac_reward_zero_std": 0.0, "grad_norm": 1.5289084911346436, "learning_rate": 9.296969696969698e-06, "loss": -0.0188, "num_tokens": 506784.0, "reward": 0.3117174506187439, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.53125, "reward_count_adherence_std": 0.25877460837364197, "reward_meter_mean": 0.6105531454086304, "reward_meter_std": 0.27458783984184265, "reward_std": 0.19874344766139984, "reward_total_composite_mean": 0.3117174506187439, "reward_total_composite_std": 0.19874346256256104, "reward_total_mean": 0.3117174506187439, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.53125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/meter/mean": 0.6105531454086304, "rewards/meter/std": 0.27458783984184265, "rewards/total_composite/mean": 0.3117174506187439, "rewards/total_composite/std": 0.19874346256256104, "sampling/importance_sampling_ratio/max": 1.9305408000946045, "sampling/importance_sampling_ratio/mean": 1.0306029319763184, "sampling/importance_sampling_ratio/min": 0.19593879580497742, "sampling/sampling_logp_difference/max": 1.6299529075622559, "sampling/sampling_logp_difference/mean": 0.12929628789424896, "step": 233 }, { "clip_ratio/high_max": 0.09212539158761501, "clip_ratio/high_mean": 0.09212539158761501, "clip_ratio/low_mean": 0.09165013581514359, "clip_ratio/low_min": 0.09165013581514359, "clip_ratio/region_mean": 0.1837755274027586, "completions/clipped_ratio": 0.0, "completions/max_length": 48.0, "completions/max_terminated_length": 48.0, "completions/mean_length": 42.25, "completions/mean_terminated_length": 42.25, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 1.387149639427662, "epoch": 0.009036144578313253, "frac_reward_zero_std": 0.0, "grad_norm": 17.917465209960938, "learning_rate": 9.293939393939395e-06, "loss": 0.1226, "num_tokens": 508394.0, "reward": 0.44727417826652527, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.44727417826652527, "reward_meter_std": 0.3157028555870056, "reward_std": 0.3157028555870056, "reward_total_composite_mean": 0.44727417826652527, "reward_total_composite_std": 0.3157028555870056, "reward_total_mean": 0.44727417826652527, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.44727417826652527, "rewards/meter/std": 0.3157028555870056, "rewards/total_composite/mean": 0.44727417826652527, "rewards/total_composite/std": 0.3157028555870056, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9895703792572021, "sampling/importance_sampling_ratio/min": 0.10507524758577347, "sampling/sampling_logp_difference/max": 2.2530784606933594, "sampling/sampling_logp_difference/mean": 0.20457275211811066, "step": 234 }, { "clip_ratio/high_max": 0.10756326653063297, "clip_ratio/high_mean": 0.10756326653063297, "clip_ratio/low_mean": 0.02009246125817299, "clip_ratio/low_min": 0.02009246125817299, "clip_ratio/region_mean": 0.12765572778880596, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 73.875, "completions/mean_terminated_length": 73.875, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 1.776310458779335, "epoch": 0.009074760580784677, "frac_reward_zero_std": 0.0, "grad_norm": 9.29736042022705, "learning_rate": 9.29090909090909e-06, "loss": 0.034, "num_tokens": 510345.0, "reward": 0.9192884564399719, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9192884564399719, "reward_meter_std": 0.1364532858133316, "reward_std": 0.1364532709121704, "reward_total_composite_mean": 0.9192884564399719, "reward_total_composite_std": 0.1364532858133316, "reward_total_mean": 0.9192884564399719, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9192884564399719, "rewards/meter/std": 0.1364532858133316, "rewards/total_composite/mean": 0.9192884564399719, "rewards/total_composite/std": 0.1364532858133316, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0462172031402588, "sampling/importance_sampling_ratio/min": 0.16506147384643555, "sampling/sampling_logp_difference/max": 1.8014373779296875, "sampling/sampling_logp_difference/mean": 0.15465568006038666, "step": 235 }, { "clip_ratio/high_max": 0.07200214825570583, "clip_ratio/high_mean": 0.07200214825570583, "clip_ratio/low_mean": 0.06175778992474079, "clip_ratio/low_min": 0.06175778992474079, "clip_ratio/region_mean": 0.13375993818044662, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 75.625, "completions/mean_terminated_length": 75.625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 1.796240046620369, "epoch": 0.0091133765832561, "frac_reward_zero_std": 0.0, "grad_norm": 8.880298614501953, "learning_rate": 9.28787878787879e-06, "loss": 0.01, "num_tokens": 512262.0, "reward": 0.6458913087844849, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6458913087844849, "reward_meter_std": 0.33680447936058044, "reward_std": 0.33680450916290283, "reward_total_composite_mean": 0.6458913087844849, "reward_total_composite_std": 0.33680447936058044, "reward_total_mean": 0.6458913087844849, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6458913087844849, "rewards/meter/std": 0.33680447936058044, "rewards/total_composite/mean": 0.6458913087844849, "rewards/total_composite/std": 0.33680447936058044, "sampling/importance_sampling_ratio/max": 1.9332317113876343, "sampling/importance_sampling_ratio/mean": 1.0316869020462036, "sampling/importance_sampling_ratio/min": 0.26908063888549805, "sampling/sampling_logp_difference/max": 1.312744140625, "sampling/sampling_logp_difference/mean": 0.14490646123886108, "step": 236 }, { "clip_ratio/high_max": 0.05680421367287636, "clip_ratio/high_mean": 0.05680421367287636, "clip_ratio/low_mean": 0.08504617214202881, "clip_ratio/low_min": 0.08504617214202881, "clip_ratio/region_mean": 0.14185038581490517, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 95.625, "completions/mean_terminated_length": 95.625, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 1.7081756442785263, "epoch": 0.009151992585727525, "frac_reward_zero_std": 0.0, "grad_norm": 8.00853443145752, "learning_rate": 9.284848484848485e-06, "loss": 0.0414, "num_tokens": 514491.0, "reward": 0.4599551856517792, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.4600679576396942, "reward_meter_std": 0.4352971911430359, "reward_std": 0.43543270230293274, "reward_total_composite_mean": 0.4599551856517792, "reward_total_composite_std": 0.43543270230293274, "reward_total_mean": 0.4599551856517792, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.4600679576396942, "rewards/meter/std": 0.4352971911430359, "rewards/total_composite/mean": 0.4599551856517792, "rewards/total_composite/std": 0.43543270230293274, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.032072901725769, "sampling/importance_sampling_ratio/min": 0.3347611427307129, "sampling/sampling_logp_difference/max": 1.0943379402160645, "sampling/sampling_logp_difference/mean": 0.1568225622177124, "step": 237 }, { "clip_ratio/high_max": 0.09207058418542147, "clip_ratio/high_mean": 0.09207058418542147, "clip_ratio/low_mean": 0.03305934276431799, "clip_ratio/low_min": 0.03305934276431799, "clip_ratio/region_mean": 0.12512992694973946, "completions/clipped_ratio": 0.0, "completions/max_length": 221.0, "completions/max_terminated_length": 221.0, "completions/mean_length": 166.25, "completions/mean_terminated_length": 166.25, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 2.2700495421886444, "epoch": 0.009190608588198949, "frac_reward_zero_std": 0.0, "grad_norm": 5.644252300262451, "learning_rate": 9.281818181818183e-06, "loss": 0.1335, "num_tokens": 517253.0, "reward": 0.5001842379570007, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.14880476891994476, "reward_meter_mean": 0.7441527843475342, "reward_meter_std": 0.38449952006340027, "reward_std": 0.37060803174972534, "reward_total_composite_mean": 0.5001842379570007, "reward_total_composite_std": 0.37060803174972534, "reward_total_mean": 0.5001842379570007, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.14880476891994476, "rewards/meter/mean": 0.7441527843475342, "rewards/meter/std": 0.38449952006340027, "rewards/total_composite/mean": 0.5001842379570007, "rewards/total_composite/std": 0.37060803174972534, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0297116041183472, "sampling/importance_sampling_ratio/min": 0.23600395023822784, "sampling/sampling_logp_difference/max": 1.4439067840576172, "sampling/sampling_logp_difference/mean": 0.16780243813991547, "step": 238 }, { "clip_ratio/high_max": 0.07056730799376965, "clip_ratio/high_mean": 0.07056730799376965, "clip_ratio/low_mean": 0.06852707732468843, "clip_ratio/low_min": 0.06852707732468843, "clip_ratio/region_mean": 0.13909438531845808, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 118.875, "completions/mean_terminated_length": 118.875, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 2.404376655817032, "epoch": 0.009229224590670373, "frac_reward_zero_std": 0.0, "grad_norm": 6.0654401779174805, "learning_rate": 9.27878787878788e-06, "loss": 0.0538, "num_tokens": 519716.0, "reward": 0.5465434193611145, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.5814238786697388, "reward_meter_std": 0.36550426483154297, "reward_std": 0.35062772035598755, "reward_total_composite_mean": 0.5465434193611145, "reward_total_composite_std": 0.35062775015830994, "reward_total_mean": 0.5465434193611145, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.5814238786697388, "rewards/meter/std": 0.36550426483154297, "rewards/total_composite/mean": 0.5465434193611145, "rewards/total_composite/std": 0.35062775015830994, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.043744683265686, "sampling/importance_sampling_ratio/min": 0.301893949508667, "sampling/sampling_logp_difference/max": 1.1976795196533203, "sampling/sampling_logp_difference/mean": 0.17544740438461304, "step": 239 }, { "clip_ratio/high_max": 0.0725616067647934, "clip_ratio/high_mean": 0.0725616067647934, "clip_ratio/low_mean": 0.04993662517517805, "clip_ratio/low_min": 0.04993662517517805, "clip_ratio/region_mean": 0.12249823193997145, "completions/clipped_ratio": 0.0, "completions/max_length": 115.0, "completions/max_terminated_length": 115.0, "completions/mean_length": 101.375, "completions/mean_terminated_length": 101.375, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 2.066037192940712, "epoch": 0.009267840593141797, "frac_reward_zero_std": 0.0, "grad_norm": 8.986166000366211, "learning_rate": 9.275757575757577e-06, "loss": 0.0204, "num_tokens": 521863.0, "reward": 0.9709064960479736, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9709064960479736, "reward_meter_std": 0.027551084756851196, "reward_std": 0.02755107544362545, "reward_total_composite_mean": 0.9709064960479736, "reward_total_composite_std": 0.027551084756851196, "reward_total_mean": 0.9709064960479736, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9709064960479736, "rewards/meter/std": 0.027551084756851196, "rewards/total_composite/mean": 0.9709064960479736, "rewards/total_composite/std": 0.027551084756851196, "sampling/importance_sampling_ratio/max": 1.8251866102218628, "sampling/importance_sampling_ratio/mean": 1.0282772779464722, "sampling/importance_sampling_ratio/min": 0.3129824995994568, "sampling/sampling_logp_difference/max": 1.1616079807281494, "sampling/sampling_logp_difference/mean": 0.16349507868289948, "step": 240 }, { "clip_ratio/high_max": 0.06964285857975483, "clip_ratio/high_mean": 0.06964285857975483, "clip_ratio/low_mean": 0.0722261555492878, "clip_ratio/low_min": 0.0722261555492878, "clip_ratio/region_mean": 0.14186901412904263, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 85.0, "completions/mean_terminated_length": 85.0, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 2.1117777675390244, "epoch": 0.009306456595613221, "frac_reward_zero_std": 0.0, "grad_norm": 8.806304931640625, "learning_rate": 9.272727272727273e-06, "loss": 0.0892, "num_tokens": 523847.0, "reward": 0.2626701593399048, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_meter_mean": 0.45606791973114014, "reward_meter_std": 0.38195329904556274, "reward_std": 0.22439199686050415, "reward_total_composite_mean": 0.2626701593399048, "reward_total_composite_std": 0.22439196705818176, "reward_total_mean": 0.2626701593399048, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/meter/mean": 0.45606791973114014, "rewards/meter/std": 0.38195329904556274, "rewards/total_composite/mean": 0.2626701593399048, "rewards/total_composite/std": 0.22439196705818176, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.031845211982727, "sampling/importance_sampling_ratio/min": 0.20610834658145905, "sampling/sampling_logp_difference/max": 1.5793533325195312, "sampling/sampling_logp_difference/mean": 0.1751883327960968, "step": 241 }, { "clip_ratio/high_max": 0.11791071016341448, "clip_ratio/high_mean": 0.11791071016341448, "clip_ratio/low_mean": 0.03575989790260792, "clip_ratio/low_min": 0.03575989790260792, "clip_ratio/region_mean": 0.1536706080660224, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 83.25, "completions/mean_terminated_length": 83.25, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 2.2592698484659195, "epoch": 0.009345072598084645, "frac_reward_zero_std": 0.0, "grad_norm": 8.190821647644043, "learning_rate": 9.26969696969697e-06, "loss": 0.0767, "num_tokens": 525777.0, "reward": 0.800595223903656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.8624944686889648, "reward_meter_std": 0.33260369300842285, "reward_std": 0.35097363591194153, "reward_total_composite_mean": 0.800595223903656, "reward_total_composite_std": 0.3509736657142639, "reward_total_mean": 0.800595223903656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.8624944686889648, "rewards/meter/std": 0.33260369300842285, "rewards/total_composite/mean": 0.800595223903656, "rewards/total_composite/std": 0.3509736657142639, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0372215509414673, "sampling/importance_sampling_ratio/min": 0.26800885796546936, "sampling/sampling_logp_difference/max": 1.3167352676391602, "sampling/sampling_logp_difference/mean": 0.16949887573719025, "step": 242 }, { "clip_ratio/high_max": 0.14678451232612133, "clip_ratio/high_mean": 0.14678451232612133, "clip_ratio/low_mean": 0.01953125, "clip_ratio/low_min": 0.01953125, "clip_ratio/region_mean": 0.16631576232612133, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 62.875, "completions/mean_terminated_length": 62.875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 1.626624509692192, "epoch": 0.009383688600556071, "frac_reward_zero_std": 0.0, "grad_norm": 9.102118492126465, "learning_rate": 9.266666666666667e-06, "loss": 0.0226, "num_tokens": 527584.0, "reward": 0.969333827495575, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.969333827495575, "reward_meter_std": 0.05763925239443779, "reward_std": 0.0576392337679863, "reward_total_composite_mean": 0.969333827495575, "reward_total_composite_std": 0.05763925239443779, "reward_total_mean": 0.969333827495575, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.969333827495575, "rewards/meter/std": 0.05763925239443779, "rewards/total_composite/mean": 0.969333827495575, "rewards/total_composite/std": 0.05763925239443779, "sampling/importance_sampling_ratio/max": 1.6970453262329102, "sampling/importance_sampling_ratio/mean": 1.0088082551956177, "sampling/importance_sampling_ratio/min": 0.2735597789287567, "sampling/sampling_logp_difference/max": 1.2962350845336914, "sampling/sampling_logp_difference/mean": 0.15184138715267181, "step": 243 }, { "clip_ratio/high_max": 0.074860580265522, "clip_ratio/high_mean": 0.074860580265522, "clip_ratio/low_mean": 0.07911759708076715, "clip_ratio/low_min": 0.07911759708076715, "clip_ratio/region_mean": 0.15397817734628916, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 118.5, "completions/mean_terminated_length": 118.5, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 2.060980185866356, "epoch": 0.009422304603027495, "frac_reward_zero_std": 0.0, "grad_norm": 7.276552677154541, "learning_rate": 9.263636363636364e-06, "loss": 0.0658, "num_tokens": 529780.0, "reward": 0.356499046087265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.3871627748012543, "reward_meter_std": 0.4140600562095642, "reward_std": 0.3705804646015167, "reward_total_composite_mean": 0.356499046087265, "reward_total_composite_std": 0.3705804646015167, "reward_total_mean": 0.356499046087265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.3871627748012543, "rewards/meter/std": 0.4140600562095642, "rewards/total_composite/mean": 0.356499046087265, "rewards/total_composite/std": 0.3705804646015167, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0257729291915894, "sampling/importance_sampling_ratio/min": 0.2939786911010742, "sampling/sampling_logp_difference/max": 1.224247932434082, "sampling/sampling_logp_difference/mean": 0.17408481240272522, "step": 244 }, { "clip_ratio/high_max": 0.05074426345527172, "clip_ratio/high_mean": 0.05074426345527172, "clip_ratio/low_mean": 0.07482307218015194, "clip_ratio/low_min": 0.07482307218015194, "clip_ratio/region_mean": 0.12556733563542366, "completions/clipped_ratio": 0.0, "completions/max_length": 139.0, "completions/max_terminated_length": 139.0, "completions/mean_length": 122.875, "completions/mean_terminated_length": 122.875, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 1.7008240818977356, "epoch": 0.00946092060549892, "frac_reward_zero_std": 0.0, "grad_norm": 7.2777557373046875, "learning_rate": 9.260606060606062e-06, "loss": -0.024, "num_tokens": 532179.0, "reward": 0.6617265939712524, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_meter_mean": 0.6986033320426941, "reward_meter_std": 0.25926199555397034, "reward_std": 0.2768361568450928, "reward_total_composite_mean": 0.6617265939712524, "reward_total_composite_std": 0.27683618664741516, "reward_total_mean": 0.6617265939712524, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/meter/mean": 0.6986033320426941, "rewards/meter/std": 0.25926199555397034, "rewards/total_composite/mean": 0.6617265939712524, "rewards/total_composite/std": 0.27683618664741516, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0323891639709473, "sampling/importance_sampling_ratio/min": 0.29009345173835754, "sampling/sampling_logp_difference/max": 1.3864390850067139, "sampling/sampling_logp_difference/mean": 0.15021459758281708, "step": 245 }, { "clip_ratio/high_max": 0.05021729413419962, "clip_ratio/high_mean": 0.05021729413419962, "clip_ratio/low_mean": 0.05107419705018401, "clip_ratio/low_min": 0.05107419705018401, "clip_ratio/region_mean": 0.10129149118438363, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 186.0, "completions/mean_length": 208.5, "completions/mean_terminated_length": 165.1428680419922, "completions/min_length": 147.0, "completions/min_terminated_length": 147.0, "entropy": 2.3770622611045837, "epoch": 0.009499536607970344, "frac_reward_zero_std": 0.0, "grad_norm": 4.300860404968262, "learning_rate": 9.257575757575759e-06, "loss": 0.0104, "num_tokens": 535079.0, "reward": 0.39897841215133667, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.16690459847450256, "reward_meter_mean": 0.622548520565033, "reward_meter_std": 0.3121766746044159, "reward_std": 0.22350633144378662, "reward_total_composite_mean": 0.39897841215133667, "reward_total_composite_std": 0.22350633144378662, "reward_total_mean": 0.39897841215133667, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.16690459847450256, "rewards/meter/mean": 0.622548520565033, "rewards/meter/std": 0.3121766746044159, "rewards/total_composite/mean": 0.39897841215133667, "rewards/total_composite/std": 0.22350633144378662, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0414752960205078, "sampling/importance_sampling_ratio/min": 0.19777174293994904, "sampling/sampling_logp_difference/max": 1.6206417083740234, "sampling/sampling_logp_difference/mean": 0.1846722960472107, "step": 246 }, { "clip_ratio/high_max": 0.13099952787160873, "clip_ratio/high_mean": 0.13099952787160873, "clip_ratio/low_mean": 0.017326733097434044, "clip_ratio/low_min": 0.017326733097434044, "clip_ratio/region_mean": 0.14832626096904278, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 2.306705191731453, "epoch": 0.009538152610441768, "frac_reward_zero_std": 0.0, "grad_norm": 9.519201278686523, "learning_rate": 9.254545454545454e-06, "loss": 0.1449, "num_tokens": 536963.0, "reward": 0.7730816006660461, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7730816006660461, "reward_meter_std": 0.314135879278183, "reward_std": 0.314135879278183, "reward_total_composite_mean": 0.7730816006660461, "reward_total_composite_std": 0.314135879278183, "reward_total_mean": 0.7730816006660461, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7730816006660461, "rewards/meter/std": 0.314135879278183, "rewards/total_composite/mean": 0.7730816006660461, "rewards/total_composite/std": 0.314135879278183, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.046431541442871, "sampling/importance_sampling_ratio/min": 0.2123212218284607, "sampling/sampling_logp_difference/max": 1.5778687000274658, "sampling/sampling_logp_difference/mean": 0.18582503497600555, "step": 247 }, { "clip_ratio/high_max": 0.08469811640679836, "clip_ratio/high_mean": 0.08469811640679836, "clip_ratio/low_mean": 0.051101832650601864, "clip_ratio/low_min": 0.051101832650601864, "clip_ratio/region_mean": 0.13579994905740023, "completions/clipped_ratio": 0.0, "completions/max_length": 244.0, "completions/max_terminated_length": 244.0, "completions/mean_length": 224.875, "completions/mean_terminated_length": 224.875, "completions/min_length": 187.0, "completions/min_terminated_length": 187.0, "entropy": 1.9894641190767288, "epoch": 0.009576768612913192, "frac_reward_zero_std": 0.0, "grad_norm": 4.550097942352295, "learning_rate": 9.251515151515152e-06, "loss": -0.0386, "num_tokens": 540434.0, "reward": 0.782259464263916, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.09300298243761063, "reward_meter_mean": 0.9458253979682922, "reward_meter_std": 0.07403269410133362, "reward_std": 0.10182006657123566, "reward_total_composite_mean": 0.782259464263916, "reward_total_composite_std": 0.10182006657123566, "reward_total_mean": 0.782259464263916, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.09300298243761063, "rewards/meter/mean": 0.9458253979682922, "rewards/meter/std": 0.07403269410133362, "rewards/total_composite/mean": 0.782259464263916, "rewards/total_composite/std": 0.10182006657123566, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0300755500793457, "sampling/importance_sampling_ratio/min": 0.23511166870594025, "sampling/sampling_logp_difference/max": 1.4476947784423828, "sampling/sampling_logp_difference/mean": 0.15468242764472961, "step": 248 }, { "clip_ratio/high_max": 0.05096696317195892, "clip_ratio/high_mean": 0.05096696317195892, "clip_ratio/low_mean": 0.08824052847921848, "clip_ratio/low_min": 0.08824052847921848, "clip_ratio/region_mean": 0.1392074916511774, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 72.625, "completions/mean_terminated_length": 72.625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 2.432593807578087, "epoch": 0.009615384615384616, "frac_reward_zero_std": 0.0, "grad_norm": 10.006518363952637, "learning_rate": 9.248484848484849e-06, "loss": 0.0622, "num_tokens": 542207.0, "reward": 0.39560720324516296, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.3977659046649933, "reward_meter_std": 0.394779771566391, "reward_std": 0.3970900774002075, "reward_total_composite_mean": 0.39560720324516296, "reward_total_composite_std": 0.3970901072025299, "reward_total_mean": 0.39560720324516296, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.3977659046649933, "rewards/meter/std": 0.394779771566391, "rewards/total_composite/mean": 0.39560720324516296, "rewards/total_composite/std": 0.3970901072025299, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0430378913879395, "sampling/importance_sampling_ratio/min": 0.1660993993282318, "sampling/sampling_logp_difference/max": 1.7951688766479492, "sampling/sampling_logp_difference/mean": 0.20861367881298065, "step": 249 }, { "clip_ratio/high_max": 0.1064140466041863, "clip_ratio/high_mean": 0.1064140466041863, "clip_ratio/low_mean": 0.03012663428671658, "clip_ratio/low_min": 0.03012663428671658, "clip_ratio/region_mean": 0.13654068089090288, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 34.75, "completions/mean_terminated_length": 34.75, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 1.6636070609092712, "epoch": 0.00965400061785604, "frac_reward_zero_std": 0.0, "grad_norm": 13.965093612670898, "learning_rate": 9.245454545454546e-06, "loss": 0.0013, "num_tokens": 543733.0, "reward": 0.7543537616729736, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7543537616729736, "reward_meter_std": 0.32807299494743347, "reward_std": 0.32807299494743347, "reward_total_composite_mean": 0.7543537616729736, "reward_total_composite_std": 0.32807299494743347, "reward_total_mean": 0.7543537616729736, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7543537616729736, "rewards/meter/std": 0.32807299494743347, "rewards/total_composite/mean": 0.7543537616729736, "rewards/total_composite/std": 0.32807299494743347, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9995890259742737, "sampling/importance_sampling_ratio/min": 0.2496013343334198, "sampling/sampling_logp_difference/max": 1.6727294921875, "sampling/sampling_logp_difference/mean": 0.17254644632339478, "step": 250 }, { "epoch": 0.00965400061785604, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.17307692307692307, "eval_completions/max_length": 494.7692307692308, "eval_completions/max_terminated_length": 335.0, "eval_completions/mean_length": 228.9903846153846, "eval_completions/mean_terminated_length": 168.23443838266226, "eval_completions/min_length": 53.38461538461539, "eval_completions/min_terminated_length": 53.38461538461539, "eval_entropy": 1.889658808708191, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 543733.0, "eval_reward": 0.3687195135996892, "eval_reward_arabic_clean_mean": 0.8942307692307693, "eval_reward_arabic_clean_std": 0.21981406670350295, "eval_reward_count_adherence_mean": 0.7803410750169021, "eval_reward_count_adherence_std": 0.24469699080173785, "eval_reward_meter_mean": 0.47416638411008394, "eval_reward_meter_std": 0.3508666604757309, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.3687195135996892, "eval_reward_total_composite_std": 0.32943402574612546, "eval_reward_total_mean": 0.3687195135996892, "eval_rewards/arabic_clean/mean": 0.8942307692307693, "eval_rewards/arabic_clean/std": 0.21981406670350295, "eval_rewards/count_adherence/mean": 0.7803410750169021, "eval_rewards/count_adherence/std": 0.24469699080173785, "eval_rewards/meter/mean": 0.47416638411008394, "eval_rewards/meter/std": 0.3508666604757309, "eval_rewards/total_composite/mean": 0.3687195135996892, "eval_rewards/total_composite/std": 0.32943402574612546, "eval_runtime": 90.6589, "eval_samples_per_second": 1.147, "eval_sampling/importance_sampling_ratio/max": 1.5637650306408222, "eval_sampling/importance_sampling_ratio/mean": 1.034668812384972, "eval_sampling/importance_sampling_ratio/min": 0.29775738372252536, "eval_sampling/sampling_logp_difference/max": 1.2276372909545898, "eval_sampling/sampling_logp_difference/mean": 0.12459980008693841, "eval_steps_per_second": 0.143, "step": 250 }, { "clip_ratio/high_max": 0.10757232829928398, "clip_ratio/high_mean": 0.10757232829928398, "clip_ratio/low_mean": 0.04646642506122589, "clip_ratio/low_min": 0.04646642506122589, "clip_ratio/region_mean": 0.15403875336050987, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 65.625, "completions/mean_terminated_length": 65.625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 2.0140078961849213, "epoch": 0.009692616620327464, "frac_reward_zero_std": 0.0, "grad_norm": 11.79161262512207, "learning_rate": 9.242424242424244e-06, "loss": 0.0119, "num_tokens": 545690.0, "reward": 0.5393182635307312, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.6017528176307678, "reward_meter_std": 0.44121459126472473, "reward_std": 0.41130462288856506, "reward_total_composite_mean": 0.5393182635307312, "reward_total_composite_std": 0.41130462288856506, "reward_total_mean": 0.5393182635307312, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.6017528176307678, "rewards/meter/std": 0.44121459126472473, "rewards/total_composite/mean": 0.5393182635307312, "rewards/total_composite/std": 0.41130462288856506, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.021461844444275, "sampling/importance_sampling_ratio/min": 0.21941626071929932, "sampling/sampling_logp_difference/max": 1.51678466796875, "sampling/sampling_logp_difference/mean": 0.1789507418870926, "step": 251 }, { "clip_ratio/high_max": 0.08756711520254612, "clip_ratio/high_mean": 0.08756711520254612, "clip_ratio/low_mean": 0.05587371252477169, "clip_ratio/low_min": 0.05587371252477169, "clip_ratio/region_mean": 0.1434408277273178, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 109.5, "completions/mean_terminated_length": 109.5, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 2.3438350409269333, "epoch": 0.009731232622798888, "frac_reward_zero_std": 0.0, "grad_norm": 6.874846935272217, "learning_rate": 9.23939393939394e-06, "loss": 0.0093, "num_tokens": 547982.0, "reward": 0.6217153668403625, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.7877767086029053, "reward_meter_std": 0.299540638923645, "reward_std": 0.2621327042579651, "reward_total_composite_mean": 0.6217153668403625, "reward_total_composite_std": 0.2621327042579651, "reward_total_mean": 0.6217153668403625, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.7877767086029053, "rewards/meter/std": 0.299540638923645, "rewards/total_composite/mean": 0.6217153668403625, "rewards/total_composite/std": 0.2621327042579651, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0337904691696167, "sampling/importance_sampling_ratio/min": 0.2464926689863205, "sampling/sampling_logp_difference/max": 1.4004230499267578, "sampling/sampling_logp_difference/mean": 0.1741628497838974, "step": 252 }, { "clip_ratio/high_max": 0.07423552963882685, "clip_ratio/high_mean": 0.07423552963882685, "clip_ratio/low_mean": 0.0667356732301414, "clip_ratio/low_min": 0.0667356732301414, "clip_ratio/region_mean": 0.14097120286896825, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 34.375, "completions/mean_terminated_length": 34.375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 2.204130381345749, "epoch": 0.009769848625270312, "frac_reward_zero_std": 0.0, "grad_norm": 14.97449016571045, "learning_rate": 9.236363636363636e-06, "loss": 0.0319, "num_tokens": 549313.0, "reward": 0.98116135597229, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.98116135597229, "reward_meter_std": 0.015678538009524345, "reward_std": 0.01567855291068554, "reward_total_composite_mean": 0.98116135597229, "reward_total_composite_std": 0.015678538009524345, "reward_total_mean": 0.98116135597229, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.98116135597229, "rewards/meter/std": 0.015678538009524345, "rewards/total_composite/mean": 0.98116135597229, "rewards/total_composite/std": 0.015678538009524345, "sampling/importance_sampling_ratio/max": 1.7639540433883667, "sampling/importance_sampling_ratio/mean": 1.0338517427444458, "sampling/importance_sampling_ratio/min": 0.2695971727371216, "sampling/sampling_logp_difference/max": 1.310826301574707, "sampling/sampling_logp_difference/mean": 0.17273853719234467, "step": 253 }, { "clip_ratio/high_max": 0.09618495963513851, "clip_ratio/high_mean": 0.09618495963513851, "clip_ratio/low_mean": 0.036308735609054565, "clip_ratio/low_min": 0.036308735609054565, "clip_ratio/region_mean": 0.13249369524419308, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 77.875, "completions/mean_terminated_length": 77.875, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 1.8999166786670685, "epoch": 0.009808464627741736, "frac_reward_zero_std": 0.0, "grad_norm": 8.44826602935791, "learning_rate": 9.233333333333334e-06, "loss": 0.0434, "num_tokens": 551128.0, "reward": 0.8029962182044983, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8029962182044983, "reward_meter_std": 0.34782952070236206, "reward_std": 0.34782955050468445, "reward_total_composite_mean": 0.8029962182044983, "reward_total_composite_std": 0.34782952070236206, "reward_total_mean": 0.8029962182044983, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8029962182044983, "rewards/meter/std": 0.34782952070236206, "rewards/total_composite/mean": 0.8029962182044983, "rewards/total_composite/std": 0.34782952070236206, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0570210218429565, "sampling/importance_sampling_ratio/min": 0.41547152400016785, "sampling/sampling_logp_difference/max": 0.89579176902771, "sampling/sampling_logp_difference/mean": 0.15848758816719055, "step": 254 }, { "clip_ratio/high_max": 0.0770185561850667, "clip_ratio/high_mean": 0.0770185561850667, "clip_ratio/low_mean": 0.03721843124367297, "clip_ratio/low_min": 0.03721843124367297, "clip_ratio/region_mean": 0.11423698742873967, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 46.0, "completions/mean_length": 98.25, "completions/mean_terminated_length": 39.142860412597656, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 2.0371719151735306, "epoch": 0.00984708063021316, "frac_reward_zero_std": 0.0, "grad_norm": 9.855134963989258, "learning_rate": 9.23030303030303e-06, "loss": 0.0437, "num_tokens": 552746.0, "reward": 0.8700155019760132, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8700155019760132, "reward_meter_std": 0.2133215367794037, "reward_std": 0.2133215218782425, "reward_total_composite_mean": 0.8700155019760132, "reward_total_composite_std": 0.2133215367794037, "reward_total_mean": 0.8700155019760132, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8700155019760132, "rewards/meter/std": 0.2133215367794037, "rewards/total_composite/mean": 0.8700155019760132, "rewards/total_composite/std": 0.2133215367794037, "sampling/importance_sampling_ratio/max": 1.803825855255127, "sampling/importance_sampling_ratio/mean": 1.0299566984176636, "sampling/importance_sampling_ratio/min": 0.29903537034988403, "sampling/sampling_logp_difference/max": 1.207193374633789, "sampling/sampling_logp_difference/mean": 0.17322853207588196, "step": 255 }, { "clip_ratio/high_max": 0.042327309027314186, "clip_ratio/high_mean": 0.042327309027314186, "clip_ratio/low_mean": 0.06572701199911535, "clip_ratio/low_min": 0.06572701199911535, "clip_ratio/region_mean": 0.10805432102642953, "completions/clipped_ratio": 0.0, "completions/max_length": 365.0, "completions/max_terminated_length": 365.0, "completions/mean_length": 162.0, "completions/mean_terminated_length": 162.0, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 1.8008338958024979, "epoch": 0.009885696632684585, "frac_reward_zero_std": 0.0, "grad_norm": 4.932265758514404, "learning_rate": 9.227272727272728e-06, "loss": 0.169, "num_tokens": 555578.0, "reward": 0.6166548728942871, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_meter_mean": 0.7288707494735718, "reward_meter_std": 0.32933905720710754, "reward_std": 0.34062808752059937, "reward_total_composite_mean": 0.6166548728942871, "reward_total_composite_std": 0.34062808752059937, "reward_total_mean": 0.6166548728942871, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/meter/mean": 0.7288707494735718, "rewards/meter/std": 0.32933905720710754, "rewards/total_composite/mean": 0.6166548728942871, "rewards/total_composite/std": 0.34062808752059937, "sampling/importance_sampling_ratio/max": 1.7278470993041992, "sampling/importance_sampling_ratio/mean": 1.0188031196594238, "sampling/importance_sampling_ratio/min": 0.2302752137184143, "sampling/sampling_logp_difference/max": 1.468480110168457, "sampling/sampling_logp_difference/mean": 0.1348879486322403, "step": 256 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.009924312635156009, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 9.224242424242424e-06, "loss": 0.0, "num_tokens": 557386.0, "reward": 0.10419389605522156, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.18269231915473938, "reward_count_adherence_std": 0.05723259970545769, "reward_meter_mean": 0.6820703744888306, "reward_meter_std": 0.41173163056373596, "reward_std": 0.10859484225511551, "reward_total_composite_mean": 0.10419389605522156, "reward_total_composite_std": 0.1085948497056961, "reward_total_mean": 0.10419389605522156, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.18269231915473938, "rewards/count_adherence/std": 0.05723259970545769, "rewards/meter/mean": 0.6820703744888306, "rewards/meter/std": 0.41173163056373596, "rewards/total_composite/mean": 0.10419389605522156, "rewards/total_composite/std": 0.1085948497056961, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 257 }, { "clip_ratio/high_max": 0.09665837325155735, "clip_ratio/high_mean": 0.09665837325155735, "clip_ratio/low_mean": 0.03242044895887375, "clip_ratio/low_min": 0.03242044895887375, "clip_ratio/region_mean": 0.1290788222104311, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 133.125, "completions/mean_terminated_length": 79.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 2.3317447006702423, "epoch": 0.009962928637627433, "frac_reward_zero_std": 0.0, "grad_norm": 4.445526123046875, "learning_rate": 9.221212121212123e-06, "loss": -0.0052, "num_tokens": 559259.0, "reward": 0.5477786660194397, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_meter_mean": 0.5733773708343506, "reward_meter_std": 0.37895193696022034, "reward_std": 0.4160669445991516, "reward_total_composite_mean": 0.5477786660194397, "reward_total_composite_std": 0.4160669445991516, "reward_total_mean": 0.5477786660194397, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/meter/mean": 0.5733773708343506, "rewards/meter/std": 0.37895193696022034, "rewards/total_composite/mean": 0.5477786660194397, "rewards/total_composite/std": 0.4160669445991516, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0478163957595825, "sampling/importance_sampling_ratio/min": 0.19116009771823883, "sampling/sampling_logp_difference/max": 1.6546440124511719, "sampling/sampling_logp_difference/mean": 0.197053000330925, "step": 258 }, { "clip_ratio/high_max": 0.12396811135113239, "clip_ratio/high_mean": 0.12396811135113239, "clip_ratio/low_mean": 0.031775956973433495, "clip_ratio/low_min": 0.031775956973433495, "clip_ratio/region_mean": 0.1557440683245659, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 64.5, "completions/mean_terminated_length": 64.5, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 2.7183807939291, "epoch": 0.010001544640098857, "frac_reward_zero_std": 0.0, "grad_norm": 11.931998252868652, "learning_rate": 9.21818181818182e-06, "loss": 0.0631, "num_tokens": 561151.0, "reward": 0.7427696585655212, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7427696585655212, "reward_meter_std": 0.428190141916275, "reward_std": 0.4281901717185974, "reward_total_composite_mean": 0.7427696585655212, "reward_total_composite_std": 0.428190141916275, "reward_total_mean": 0.7427696585655212, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7427696585655212, "rewards/meter/std": 0.428190141916275, "rewards/total_composite/mean": 0.7427696585655212, "rewards/total_composite/std": 0.428190141916275, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0478657484054565, "sampling/importance_sampling_ratio/min": 0.248060405254364, "sampling/sampling_logp_difference/max": 1.394083023071289, "sampling/sampling_logp_difference/mean": 0.19922052323818207, "step": 259 }, { "clip_ratio/high_max": 0.09239057265222073, "clip_ratio/high_mean": 0.09239057265222073, "clip_ratio/low_mean": 0.04702932108193636, "clip_ratio/low_min": 0.04702932108193636, "clip_ratio/region_mean": 0.13941989373415709, "completions/clipped_ratio": 0.0, "completions/max_length": 100.0, "completions/max_terminated_length": 100.0, "completions/mean_length": 78.0, "completions/mean_terminated_length": 78.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 2.69622403383255, "epoch": 0.010040160642570281, "frac_reward_zero_std": 0.0, "grad_norm": 8.251180648803711, "learning_rate": 9.215151515151515e-06, "loss": 0.0645, "num_tokens": 563063.0, "reward": 0.7421708106994629, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9256542325019836, "reward_meter_std": 0.16513100266456604, "reward_std": 0.3694427013397217, "reward_total_composite_mean": 0.7421708106994629, "reward_total_composite_std": 0.3694427013397217, "reward_total_mean": 0.7421708106994629, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9256542325019836, "rewards/meter/std": 0.16513100266456604, "rewards/total_composite/mean": 0.7421708106994629, "rewards/total_composite/std": 0.3694427013397217, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0489428043365479, "sampling/importance_sampling_ratio/min": 0.15505114197731018, "sampling/sampling_logp_difference/max": 1.8640003204345703, "sampling/sampling_logp_difference/mean": 0.18370838463306427, "step": 260 }, { "clip_ratio/high_max": 0.18774798419326544, "clip_ratio/high_mean": 0.18774798419326544, "clip_ratio/low_mean": 0.023148147389292717, "clip_ratio/low_min": 0.023148147389292717, "clip_ratio/region_mean": 0.21089613158255816, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 34.25, "completions/mean_terminated_length": 34.25, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 2.57423797249794, "epoch": 0.010078776645041705, "frac_reward_zero_std": 0.0, "grad_norm": 23.678274154663086, "learning_rate": 9.212121212121213e-06, "loss": -0.0436, "num_tokens": 564601.0, "reward": 0.9117841720581055, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9117841720581055, "reward_meter_std": 0.18696558475494385, "reward_std": 0.18696556985378265, "reward_total_composite_mean": 0.9117841720581055, "reward_total_composite_std": 0.18696558475494385, "reward_total_mean": 0.9117841720581055, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9117841720581055, "rewards/meter/std": 0.18696558475494385, "rewards/total_composite/mean": 0.9117841720581055, "rewards/total_composite/std": 0.18696558475494385, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0213043689727783, "sampling/importance_sampling_ratio/min": 0.2197856903076172, "sampling/sampling_logp_difference/max": 1.5957603454589844, "sampling/sampling_logp_difference/mean": 0.20806393027305603, "step": 261 }, { "clip_ratio/high_max": 0.09604987595230341, "clip_ratio/high_mean": 0.09604987595230341, "clip_ratio/low_mean": 0.04403977282345295, "clip_ratio/low_min": 0.04403977282345295, "clip_ratio/region_mean": 0.14008964877575636, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 67.875, "completions/mean_terminated_length": 67.875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 2.2145213186740875, "epoch": 0.01011739264751313, "frac_reward_zero_std": 0.0, "grad_norm": 8.469246864318848, "learning_rate": 9.20909090909091e-06, "loss": 0.027, "num_tokens": 566352.0, "reward": 0.746609091758728, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7471080422401428, "reward_meter_std": 0.37732023000717163, "reward_std": 0.37844428420066833, "reward_total_composite_mean": 0.746609091758728, "reward_total_composite_std": 0.3784443438053131, "reward_total_mean": 0.746609091758728, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7471080422401428, "rewards/meter/std": 0.37732023000717163, "rewards/total_composite/mean": 0.746609091758728, "rewards/total_composite/std": 0.3784443438053131, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0465917587280273, "sampling/importance_sampling_ratio/min": 0.2786925435066223, "sampling/sampling_logp_difference/max": 1.2776460647583008, "sampling/sampling_logp_difference/mean": 0.180844247341156, "step": 262 }, { "clip_ratio/high_max": 0.022365196608006954, "clip_ratio/high_mean": 0.022365196608006954, "clip_ratio/low_mean": 0.007675438653677702, "clip_ratio/low_min": 0.007675438653677702, "clip_ratio/region_mean": 0.030040635261684656, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 364.75, "completions/mean_terminated_length": 119.33333587646484, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.768448993563652, "epoch": 0.010156008649984553, "frac_reward_zero_std": 0.0, "grad_norm": 1.5207958221435547, "learning_rate": 9.206060606060607e-06, "loss": -0.0374, "num_tokens": 568030.0, "reward": 0.4171389043331146, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.18898223340511322, "reward_meter_mean": 0.548949122428894, "reward_meter_std": 0.3854943811893463, "reward_std": 0.4122898578643799, "reward_total_composite_mean": 0.4171389043331146, "reward_total_composite_std": 0.4122898578643799, "reward_total_mean": 0.4171389043331146, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.18898223340511322, "rewards/meter/mean": 0.548949122428894, "rewards/meter/std": 0.3854943811893463, "rewards/total_composite/mean": 0.4171389043331146, "rewards/total_composite/std": 0.4122898578643799, "sampling/importance_sampling_ratio/max": 1.990702748298645, "sampling/importance_sampling_ratio/mean": 1.0653656721115112, "sampling/importance_sampling_ratio/min": 0.43710729479789734, "sampling/sampling_logp_difference/max": 0.8275766372680664, "sampling/sampling_logp_difference/mean": 0.14995871484279633, "step": 263 }, { "clip_ratio/high_max": 0.052887228317558765, "clip_ratio/high_mean": 0.052887228317558765, "clip_ratio/low_mean": 0.04904576577246189, "clip_ratio/low_min": 0.04904576577246189, "clip_ratio/region_mean": 0.10193299409002066, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 172.0, "completions/mean_length": 190.0, "completions/mean_terminated_length": 144.0, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 1.793292224407196, "epoch": 0.010194624652455977, "frac_reward_zero_std": 0.0, "grad_norm": 3.842879056930542, "learning_rate": 9.203030303030304e-06, "loss": 0.0244, "num_tokens": 570502.0, "reward": 0.2761991024017334, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.18898223340511322, "reward_meter_mean": 0.3928939700126648, "reward_meter_std": 0.41266870498657227, "reward_std": 0.37751540541648865, "reward_total_composite_mean": 0.2761991024017334, "reward_total_composite_std": 0.37751540541648865, "reward_total_mean": 0.2761991024017334, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.18898223340511322, "rewards/meter/mean": 0.3928939700126648, "rewards/meter/std": 0.41266870498657227, "rewards/total_composite/mean": 0.2761991024017334, "rewards/total_composite/std": 0.37751540541648865, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.032580018043518, "sampling/importance_sampling_ratio/min": 0.2537790536880493, "sampling/sampling_logp_difference/max": 1.371291160583496, "sampling/sampling_logp_difference/mean": 0.15981298685073853, "step": 264 }, { "clip_ratio/high_max": 0.05313897877931595, "clip_ratio/high_mean": 0.05313897877931595, "clip_ratio/low_mean": 0.047125319950282574, "clip_ratio/low_min": 0.047125319950282574, "clip_ratio/region_mean": 0.10026429872959852, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 184.25, "completions/mean_terminated_length": 75.0, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 1.5678513646125793, "epoch": 0.010233240654927402, "frac_reward_zero_std": 0.0, "grad_norm": 3.180842876434326, "learning_rate": 9.200000000000002e-06, "loss": -0.0242, "num_tokens": 572168.0, "reward": 0.3810850977897644, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_meter_mean": 0.3811070919036865, "reward_meter_std": 0.4830027222633362, "reward_std": 0.483022540807724, "reward_total_composite_mean": 0.3810850977897644, "reward_total_composite_std": 0.483022540807724, "reward_total_mean": 0.3810850977897644, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/meter/mean": 0.3811070919036865, "rewards/meter/std": 0.4830027222633362, "rewards/total_composite/mean": 0.3810850977897644, "rewards/total_composite/std": 0.483022540807724, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0414031744003296, "sampling/importance_sampling_ratio/min": 0.30413204431533813, "sampling/sampling_logp_difference/max": 1.190293312072754, "sampling/sampling_logp_difference/mean": 0.17107190191745758, "step": 265 }, { "clip_ratio/high_max": 0.021464127115905285, "clip_ratio/high_mean": 0.021464127115905285, "clip_ratio/low_mean": 0.01574307307600975, "clip_ratio/low_min": 0.01574307307600975, "clip_ratio/region_mean": 0.037207200191915035, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 452.5, "completions/mean_terminated_length": 353.3333435058594, "completions/min_length": 309.0, "completions/min_terminated_length": 309.0, "entropy": 0.6620796918869019, "epoch": 0.010271856657398826, "frac_reward_zero_std": 0.0, "grad_norm": 2.051448345184326, "learning_rate": 9.196969696969697e-06, "loss": -0.2009, "num_tokens": 574908.0, "reward": 0.14004071056842804, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.1836577206850052, "reward_meter_mean": 0.2717985510826111, "reward_meter_std": 0.30201706290245056, "reward_std": 0.2556873857975006, "reward_total_composite_mean": 0.14004071056842804, "reward_total_composite_std": 0.255687415599823, "reward_total_mean": 0.14004071056842804, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.1836577206850052, "rewards/meter/mean": 0.2717985510826111, "rewards/meter/std": 0.30201706290245056, "rewards/total_composite/mean": 0.14004071056842804, "rewards/total_composite/std": 0.255687415599823, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0511150360107422, "sampling/importance_sampling_ratio/min": 0.3106938600540161, "sampling/sampling_logp_difference/max": 1.1689472198486328, "sampling/sampling_logp_difference/mean": 0.15083111822605133, "step": 266 }, { "clip_ratio/high_max": 0.12783648911863565, "clip_ratio/high_mean": 0.12783648911863565, "clip_ratio/low_mean": 0.04466869868338108, "clip_ratio/low_min": 0.04466869868338108, "clip_ratio/region_mean": 0.17250518780201674, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 70.5, "completions/mean_terminated_length": 70.5, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 2.6592743396759033, "epoch": 0.01031047265987025, "frac_reward_zero_std": 0.0, "grad_norm": 10.749847412109375, "learning_rate": 9.193939393939395e-06, "loss": 0.1645, "num_tokens": 576616.0, "reward": 0.7857815027236938, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7857816219329834, "reward_meter_std": 0.35958775877952576, "reward_std": 0.35958805680274963, "reward_total_composite_mean": 0.7857815027236938, "reward_total_composite_std": 0.359588086605072, "reward_total_mean": 0.7857815027236938, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7857816219329834, "rewards/meter/std": 0.35958775877952576, "rewards/total_composite/mean": 0.7857815027236938, "rewards/total_composite/std": 0.359588086605072, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.053087830543518, "sampling/importance_sampling_ratio/min": 0.2771447002887726, "sampling/sampling_logp_difference/max": 1.2832155227661133, "sampling/sampling_logp_difference/mean": 0.19864748418331146, "step": 267 }, { "clip_ratio/high_max": 0.1120001059025526, "clip_ratio/high_mean": 0.1120001059025526, "clip_ratio/low_mean": 0.06932727806270123, "clip_ratio/low_min": 0.06932727806270123, "clip_ratio/region_mean": 0.18132738396525383, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 51.375, "completions/mean_terminated_length": 51.375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 1.0063387379050255, "epoch": 0.010349088662341674, "frac_reward_zero_std": 0.0, "grad_norm": 18.469411849975586, "learning_rate": 9.190909090909092e-06, "loss": 0.0142, "num_tokens": 578371.0, "reward": 0.6911875009536743, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6911875009536743, "reward_meter_std": 0.36712267994880676, "reward_std": 0.3671226501464844, "reward_total_composite_mean": 0.6911875009536743, "reward_total_composite_std": 0.36712267994880676, "reward_total_mean": 0.6911875009536743, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6911875009536743, "rewards/meter/std": 0.36712267994880676, "rewards/total_composite/mean": 0.6911875009536743, "rewards/total_composite/std": 0.36712267994880676, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.980422854423523, "sampling/importance_sampling_ratio/min": 0.04933354631066322, "sampling/sampling_logp_difference/max": 3.009150981903076, "sampling/sampling_logp_difference/mean": 0.23253513872623444, "step": 268 }, { "clip_ratio/high_max": 0.09056044183671474, "clip_ratio/high_mean": 0.09056044183671474, "clip_ratio/low_mean": 0.049313412979245186, "clip_ratio/low_min": 0.049313412979245186, "clip_ratio/region_mean": 0.13987385481595993, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 92.75, "completions/mean_terminated_length": 92.75, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 2.058430328965187, "epoch": 0.010387704664813098, "frac_reward_zero_std": 0.0, "grad_norm": 7.486761569976807, "learning_rate": 9.187878787878789e-06, "loss": 0.0128, "num_tokens": 580441.0, "reward": 0.7411905527114868, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8626927137374878, "reward_meter_std": 0.2941436171531677, "reward_std": 0.41744592785835266, "reward_total_composite_mean": 0.7411905527114868, "reward_total_composite_std": 0.41744595766067505, "reward_total_mean": 0.7411905527114868, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8626927137374878, "rewards/meter/std": 0.2941436171531677, "rewards/total_composite/mean": 0.7411905527114868, "rewards/total_composite/std": 0.41744595766067505, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0476194620132446, "sampling/importance_sampling_ratio/min": 0.17622657120227814, "sampling/sampling_logp_difference/max": 1.7359848022460938, "sampling/sampling_logp_difference/mean": 0.1645209938287735, "step": 269 }, { "clip_ratio/high_max": 0.019736841320991516, "clip_ratio/high_mean": 0.019736841320991516, "clip_ratio/low_mean": 0.013358778320252895, "clip_ratio/low_min": 0.013358778320252895, "clip_ratio/region_mean": 0.03309561964124441, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 262.0, "completions/mean_length": 445.25, "completions/mean_terminated_length": 245.0, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.48567473888397217, "epoch": 0.010426320667284522, "frac_reward_zero_std": 0.0, "grad_norm": 1.4978303909301758, "learning_rate": 9.184848484848485e-06, "loss": -0.0971, "num_tokens": 582411.0, "reward": 0.030055876821279526, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.1157275140285492, "reward_meter_mean": 0.07851605117321014, "reward_meter_std": 0.06109999120235443, "reward_std": 0.040369123220443726, "reward_total_composite_mean": 0.030055876821279526, "reward_total_composite_std": 0.040369123220443726, "reward_total_mean": 0.030055876821279526, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.1157275140285492, "rewards/meter/mean": 0.07851605117321014, "rewards/meter/std": 0.06109999120235443, "rewards/total_composite/mean": 0.030055876821279526, "rewards/total_composite/std": 0.040369123220443726, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0170917510986328, "sampling/importance_sampling_ratio/min": 0.22663357853889465, "sampling/sampling_logp_difference/max": 1.4844207763671875, "sampling/sampling_logp_difference/mean": 0.1704261153936386, "step": 270 }, { "clip_ratio/high_max": 0.09404239989817142, "clip_ratio/high_mean": 0.09404239989817142, "clip_ratio/low_mean": 0.07794231548905373, "clip_ratio/low_min": 0.07794231548905373, "clip_ratio/region_mean": 0.17198471538722515, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 59.0, "completions/mean_terminated_length": 59.0, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 1.7888787239789963, "epoch": 0.010464936669755946, "frac_reward_zero_std": 0.0, "grad_norm": 10.39463996887207, "learning_rate": 9.181818181818184e-06, "loss": 0.0604, "num_tokens": 584259.0, "reward": 0.6009393930435181, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6009393930435181, "reward_meter_std": 0.41454699635505676, "reward_std": 0.41454702615737915, "reward_total_composite_mean": 0.6009393930435181, "reward_total_composite_std": 0.41454699635505676, "reward_total_mean": 0.6009393930435181, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6009393930435181, "rewards/meter/std": 0.41454699635505676, "rewards/total_composite/mean": 0.6009393930435181, "rewards/total_composite/std": 0.41454699635505676, "sampling/importance_sampling_ratio/max": 1.8810900449752808, "sampling/importance_sampling_ratio/mean": 1.017964482307434, "sampling/importance_sampling_ratio/min": 0.3745101988315582, "sampling/sampling_logp_difference/max": 0.9821362495422363, "sampling/sampling_logp_difference/mean": 0.16193664073944092, "step": 271 }, { "clip_ratio/high_max": 0.07653909968212247, "clip_ratio/high_mean": 0.07653909968212247, "clip_ratio/low_mean": 0.059118627570569515, "clip_ratio/low_min": 0.059118627570569515, "clip_ratio/region_mean": 0.13565772725269198, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 37.625, "completions/mean_terminated_length": 37.625, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 2.1982433944940567, "epoch": 0.01050355267222737, "frac_reward_zero_std": 0.0, "grad_norm": 16.354007720947266, "learning_rate": 9.178787878787879e-06, "loss": 0.239, "num_tokens": 585848.0, "reward": 0.5806645154953003, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.6360828876495361, "reward_meter_std": 0.43516474962234497, "reward_std": 0.4882129728794098, "reward_total_composite_mean": 0.5806645154953003, "reward_total_composite_std": 0.4882129728794098, "reward_total_mean": 0.5806645154953003, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.6360828876495361, "rewards/meter/std": 0.43516474962234497, "rewards/total_composite/mean": 0.5806645154953003, "rewards/total_composite/std": 0.4882129728794098, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0565941333770752, "sampling/importance_sampling_ratio/min": 0.19141019880771637, "sampling/sampling_logp_difference/max": 1.653336524963379, "sampling/sampling_logp_difference/mean": 0.1913789063692093, "step": 272 }, { "clip_ratio/high_max": 0.006777108646929264, "clip_ratio/high_mean": 0.006777108646929264, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006777108646929264, "completions/clipped_ratio": 0.875, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 510.25, "completions/mean_terminated_length": 498.0, "completions/min_length": 498.0, "completions/min_terminated_length": 498.0, "entropy": 0.12209701538085938, "epoch": 0.010542168674698794, "frac_reward_zero_std": 0.0, "grad_norm": 1.6592974662780762, "learning_rate": 9.175757575757576e-06, "loss": -0.2266, "num_tokens": 588010.0, "reward": 0.11888930201530457, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_count_adherence_mean": 0.4821428656578064, "reward_count_adherence_std": 0.23458294570446014, "reward_meter_mean": 0.31048354506492615, "reward_meter_std": 0.3288111686706543, "reward_std": 0.20482073724269867, "reward_total_composite_mean": 0.11888930201530457, "reward_total_composite_std": 0.20482075214385986, "reward_total_mean": 0.11888930201530457, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/count_adherence/mean": 0.4821428656578064, "rewards/count_adherence/std": 0.23458294570446014, "rewards/meter/mean": 0.31048354506492615, "rewards/meter/std": 0.3288111686706543, "rewards/total_composite/mean": 0.11888930201530457, "rewards/total_composite/std": 0.20482075214385986, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0267904996871948, "sampling/importance_sampling_ratio/min": 0.3580196797847748, "sampling/sampling_logp_difference/max": 1.0271673202514648, "sampling/sampling_logp_difference/mean": 0.08662550151348114, "step": 273 }, { "clip_ratio/high_max": 0.06644443050026894, "clip_ratio/high_mean": 0.06644443050026894, "clip_ratio/low_mean": 0.024902896489948034, "clip_ratio/low_min": 0.024902896489948034, "clip_ratio/region_mean": 0.09134732699021697, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 213.0, "completions/mean_length": 229.75, "completions/mean_terminated_length": 189.42857360839844, "completions/min_length": 160.0, "completions/min_terminated_length": 160.0, "entropy": 1.431438848376274, "epoch": 0.010580784677170219, "frac_reward_zero_std": 0.0, "grad_norm": 3.6858744621276855, "learning_rate": 9.172727272727274e-06, "loss": -0.1042, "num_tokens": 590872.0, "reward": 0.39238011837005615, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8035714626312256, "reward_count_adherence_std": 0.15152288973331451, "reward_meter_mean": 0.46854132413864136, "reward_meter_std": 0.38436928391456604, "reward_std": 0.34078967571258545, "reward_total_composite_mean": 0.39238011837005615, "reward_total_composite_std": 0.34078967571258545, "reward_total_mean": 0.39238011837005615, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8035714626312256, "rewards/count_adherence/std": 0.15152288973331451, "rewards/meter/mean": 0.46854132413864136, "rewards/meter/std": 0.38436928391456604, "rewards/total_composite/mean": 0.39238011837005615, "rewards/total_composite/std": 0.34078967571258545, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0281857252120972, "sampling/importance_sampling_ratio/min": 0.3060767352581024, "sampling/sampling_logp_difference/max": 1.1839194297790527, "sampling/sampling_logp_difference/mean": 0.14069926738739014, "step": 274 }, { "clip_ratio/high_max": 0.0879957852885127, "clip_ratio/high_mean": 0.0879957852885127, "clip_ratio/low_mean": 0.017326733097434044, "clip_ratio/low_min": 0.017326733097434044, "clip_ratio/region_mean": 0.10532251838594675, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 198.625, "completions/mean_terminated_length": 94.16667175292969, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 1.5135751068592072, "epoch": 0.010619400679641643, "frac_reward_zero_std": 0.0, "grad_norm": 3.059723138809204, "learning_rate": 9.169696969696971e-06, "loss": -0.1344, "num_tokens": 592749.0, "reward": 0.5918634533882141, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.6977724432945251, "reward_meter_std": 0.38409724831581116, "reward_std": 0.4484126567840576, "reward_total_composite_mean": 0.5918634533882141, "reward_total_composite_std": 0.44841268658638, "reward_total_mean": 0.5918634533882141, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.6977724432945251, "rewards/meter/std": 0.38409724831581116, "rewards/total_composite/mean": 0.5918634533882141, "rewards/total_composite/std": 0.44841268658638, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.050614833831787, "sampling/importance_sampling_ratio/min": 0.2916625738143921, "sampling/sampling_logp_difference/max": 1.2321577072143555, "sampling/sampling_logp_difference/mean": 0.17513571679592133, "step": 275 }, { "clip_ratio/high_max": 0.06428467482328415, "clip_ratio/high_mean": 0.06428467482328415, "clip_ratio/low_mean": 0.05223934445530176, "clip_ratio/low_min": 0.05223934445530176, "clip_ratio/region_mean": 0.11652401927858591, "completions/clipped_ratio": 0.0, "completions/max_length": 148.0, "completions/max_terminated_length": 148.0, "completions/mean_length": 116.125, "completions/mean_terminated_length": 116.125, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 2.8212269842624664, "epoch": 0.010658016682113068, "frac_reward_zero_std": 0.0, "grad_norm": 6.9882683753967285, "learning_rate": 9.166666666666666e-06, "loss": 0.062, "num_tokens": 594942.0, "reward": 0.47085532546043396, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.8023958206176758, "reward_meter_std": 0.32191282510757446, "reward_std": 0.46588125824928284, "reward_total_composite_mean": 0.47085532546043396, "reward_total_composite_std": 0.4658812880516052, "reward_total_mean": 0.47085532546043396, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.8023958206176758, "rewards/meter/std": 0.32191282510757446, "rewards/total_composite/mean": 0.47085532546043396, "rewards/total_composite/std": 0.4658812880516052, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0579421520233154, "sampling/importance_sampling_ratio/min": 0.21824216842651367, "sampling/sampling_logp_difference/max": 1.5221500396728516, "sampling/sampling_logp_difference/mean": 0.18583007156848907, "step": 276 }, { "clip_ratio/high_max": 0.12392107397317886, "clip_ratio/high_mean": 0.12392107397317886, "clip_ratio/low_mean": 0.04722222313284874, "clip_ratio/low_min": 0.04722222313284874, "clip_ratio/region_mean": 0.1711432971060276, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 46.375, "completions/mean_terminated_length": 46.375, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 1.2117373794317245, "epoch": 0.010696632684584493, "frac_reward_zero_std": 0.0, "grad_norm": 16.393516540527344, "learning_rate": 9.163636363636365e-06, "loss": 0.0701, "num_tokens": 596617.0, "reward": 0.5784964561462402, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5784964561462402, "reward_meter_std": 0.3392086327075958, "reward_std": 0.3392086327075958, "reward_total_composite_mean": 0.5784964561462402, "reward_total_composite_std": 0.3392086327075958, "reward_total_mean": 0.5784964561462402, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5784964561462402, "rewards/meter/std": 0.3392086327075958, "rewards/total_composite/mean": 0.5784964561462402, "rewards/total_composite/std": 0.3392086327075958, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0258041620254517, "sampling/importance_sampling_ratio/min": 0.12757723033428192, "sampling/sampling_logp_difference/max": 2.0590333938598633, "sampling/sampling_logp_difference/mean": 0.2193935364484787, "step": 277 }, { "clip_ratio/high_max": 0.0827424954622984, "clip_ratio/high_mean": 0.0827424954622984, "clip_ratio/low_mean": 0.03383508883416653, "clip_ratio/low_min": 0.03383508883416653, "clip_ratio/region_mean": 0.11657758429646492, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 111.625, "completions/mean_terminated_length": 111.625, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 1.7266891598701477, "epoch": 0.010735248687055917, "frac_reward_zero_std": 0.0, "grad_norm": 7.285430431365967, "learning_rate": 9.160606060606061e-06, "loss": 0.0379, "num_tokens": 598870.0, "reward": 0.8526768684387207, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8526768684387207, "reward_meter_std": 0.27573806047439575, "reward_std": 0.27573806047439575, "reward_total_composite_mean": 0.8526768684387207, "reward_total_composite_std": 0.27573806047439575, "reward_total_mean": 0.8526768684387207, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8526768684387207, "rewards/meter/std": 0.27573806047439575, "rewards/total_composite/mean": 0.8526768684387207, "rewards/total_composite/std": 0.27573806047439575, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0367151498794556, "sampling/importance_sampling_ratio/min": 0.2770197093486786, "sampling/sampling_logp_difference/max": 1.2836666107177734, "sampling/sampling_logp_difference/mean": 0.14818187057971954, "step": 278 }, { "clip_ratio/high_max": 0.05927513726055622, "clip_ratio/high_mean": 0.05927513726055622, "clip_ratio/low_mean": 0.08847619779407978, "clip_ratio/low_min": 0.08847619779407978, "clip_ratio/region_mean": 0.147751335054636, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 59.875, "completions/mean_terminated_length": 59.875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 1.8726500868797302, "epoch": 0.01077386468952734, "frac_reward_zero_std": 0.0, "grad_norm": 11.760062217712402, "learning_rate": 9.157575757575758e-06, "loss": 0.0423, "num_tokens": 600549.0, "reward": 0.45295605063438416, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.45295605063438416, "reward_meter_std": 0.416568398475647, "reward_std": 0.416568398475647, "reward_total_composite_mean": 0.45295605063438416, "reward_total_composite_std": 0.416568398475647, "reward_total_mean": 0.45295605063438416, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.45295605063438416, "rewards/meter/std": 0.416568398475647, "rewards/total_composite/mean": 0.45295605063438416, "rewards/total_composite/std": 0.416568398475647, "sampling/importance_sampling_ratio/max": 1.9151904582977295, "sampling/importance_sampling_ratio/mean": 1.029968023300171, "sampling/importance_sampling_ratio/min": 0.35172298550605774, "sampling/sampling_logp_difference/max": 1.0449113845825195, "sampling/sampling_logp_difference/mean": 0.16330303251743317, "step": 279 }, { "clip_ratio/high_max": 0.08833360858261585, "clip_ratio/high_mean": 0.08833360858261585, "clip_ratio/low_mean": 0.06699808966368437, "clip_ratio/low_min": 0.06699808966368437, "clip_ratio/region_mean": 0.15533169824630022, "completions/clipped_ratio": 0.0, "completions/max_length": 187.0, "completions/max_terminated_length": 187.0, "completions/mean_length": 154.0, "completions/mean_terminated_length": 154.0, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 1.5907130688428879, "epoch": 0.010812480691998765, "frac_reward_zero_std": 0.0, "grad_norm": 7.997870922088623, "learning_rate": 9.154545454545455e-06, "loss": 0.0106, "num_tokens": 603365.0, "reward": 0.404763400554657, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8958333134651184, "reward_count_adherence_std": 0.08625820279121399, "reward_meter_mean": 0.4514502286911011, "reward_meter_std": 0.3197373151779175, "reward_std": 0.32716503739356995, "reward_total_composite_mean": 0.404763400554657, "reward_total_composite_std": 0.32716503739356995, "reward_total_mean": 0.404763400554657, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8958333134651184, "rewards/count_adherence/std": 0.08625820279121399, "rewards/meter/mean": 0.4514502286911011, "rewards/meter/std": 0.3197373151779175, "rewards/total_composite/mean": 0.404763400554657, "rewards/total_composite/std": 0.32716503739356995, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0190558433532715, "sampling/importance_sampling_ratio/min": 0.19574791193008423, "sampling/sampling_logp_difference/max": 1.630927562713623, "sampling/sampling_logp_difference/mean": 0.1659134477376938, "step": 280 }, { "clip_ratio/high_max": 0.07967114355415106, "clip_ratio/high_mean": 0.07967114355415106, "clip_ratio/low_mean": 0.02150537632405758, "clip_ratio/low_min": 0.02150537632405758, "clip_ratio/region_mean": 0.10117651987820864, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 205.0, "completions/mean_terminated_length": 102.66667175292969, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 1.8442810326814651, "epoch": 0.010851096694470189, "frac_reward_zero_std": 0.0, "grad_norm": 1.4875062704086304, "learning_rate": 9.151515151515153e-06, "loss": -0.1226, "num_tokens": 605333.0, "reward": 0.8265775442123413, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.8328183889389038, "reward_meter_std": 0.3320009410381317, "reward_std": 0.34886112809181213, "reward_total_composite_mean": 0.8265775442123413, "reward_total_composite_std": 0.3488611578941345, "reward_total_mean": 0.8265775442123413, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.8328183889389038, "rewards/meter/std": 0.3320009410381317, "rewards/total_composite/mean": 0.8265775442123413, "rewards/total_composite/std": 0.3488611578941345, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0586100816726685, "sampling/importance_sampling_ratio/min": 0.16476187109947205, "sampling/sampling_logp_difference/max": 1.8032541275024414, "sampling/sampling_logp_difference/mean": 0.17393165826797485, "step": 281 }, { "clip_ratio/high_max": 0.05353208538144827, "clip_ratio/high_mean": 0.05353208538144827, "clip_ratio/low_mean": 0.023917971178889275, "clip_ratio/low_min": 0.023917971178889275, "clip_ratio/region_mean": 0.07745005656033754, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 193.0, "completions/mean_length": 308.125, "completions/mean_terminated_length": 185.8000030517578, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 1.0763673335313797, "epoch": 0.010889712696941613, "frac_reward_zero_std": 0.0, "grad_norm": 2.2956902980804443, "learning_rate": 9.148484848484848e-06, "loss": -0.2053, "num_tokens": 607958.0, "reward": 0.27106142044067383, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.15430334210395813, "reward_meter_mean": 0.3302040994167328, "reward_meter_std": 0.34381669759750366, "reward_std": 0.29077771306037903, "reward_total_composite_mean": 0.27106142044067383, "reward_total_composite_std": 0.29077771306037903, "reward_total_mean": 0.27106142044067383, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.15430334210395813, "rewards/meter/mean": 0.3302040994167328, "rewards/meter/std": 0.34381669759750366, "rewards/total_composite/mean": 0.27106142044067383, "rewards/total_composite/std": 0.29077771306037903, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0214406251907349, "sampling/importance_sampling_ratio/min": 0.1493440419435501, "sampling/sampling_logp_difference/max": 1.9015026092529297, "sampling/sampling_logp_difference/mean": 0.1307448148727417, "step": 282 }, { "clip_ratio/high_max": 0.11775326356291771, "clip_ratio/high_mean": 0.11775326356291771, "clip_ratio/low_mean": 0.01697530783712864, "clip_ratio/low_min": 0.01697530783712864, "clip_ratio/region_mean": 0.13472857140004635, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 140.375, "completions/mean_terminated_length": 140.375, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 1.7298996597528458, "epoch": 0.010928328699413037, "frac_reward_zero_std": 0.0, "grad_norm": 6.667464256286621, "learning_rate": 9.145454545454546e-06, "loss": 0.0578, "num_tokens": 610721.0, "reward": 0.7411341667175293, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.8891458511352539, "reward_meter_std": 0.24647405743598938, "reward_std": 0.1899271160364151, "reward_total_composite_mean": 0.7411341667175293, "reward_total_composite_std": 0.1899271160364151, "reward_total_mean": 0.7411341667175293, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.8891458511352539, "rewards/meter/std": 0.24647405743598938, "rewards/total_composite/mean": 0.7411341667175293, "rewards/total_composite/std": 0.1899271160364151, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0217300653457642, "sampling/importance_sampling_ratio/min": 0.2201007455587387, "sampling/sampling_logp_difference/max": 1.6330780982971191, "sampling/sampling_logp_difference/mean": 0.15236106514930725, "step": 283 }, { "clip_ratio/high_max": 0.09173304960131645, "clip_ratio/high_mean": 0.09173304960131645, "clip_ratio/low_mean": 0.03779107145965099, "clip_ratio/low_min": 0.03779107145965099, "clip_ratio/region_mean": 0.12952412106096745, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 96.125, "completions/mean_terminated_length": 96.125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 1.9220721423625946, "epoch": 0.010966944701884461, "frac_reward_zero_std": 0.0, "grad_norm": 7.729722499847412, "learning_rate": 9.142424242424243e-06, "loss": 0.0181, "num_tokens": 612706.0, "reward": 0.9732859134674072, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9732859134674072, "reward_meter_std": 0.022573497146368027, "reward_std": 0.02257349155843258, "reward_total_composite_mean": 0.9732859134674072, "reward_total_composite_std": 0.022573497146368027, "reward_total_mean": 0.9732859134674072, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9732859134674072, "rewards/meter/std": 0.022573497146368027, "rewards/total_composite/mean": 0.9732859134674072, "rewards/total_composite/std": 0.022573497146368027, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.028389573097229, "sampling/importance_sampling_ratio/min": 0.3058738112449646, "sampling/sampling_logp_difference/max": 1.1845827102661133, "sampling/sampling_logp_difference/mean": 0.14632542431354523, "step": 284 }, { "clip_ratio/high_max": 0.05674981512129307, "clip_ratio/high_mean": 0.05674981512129307, "clip_ratio/low_mean": 0.0852982671931386, "clip_ratio/low_min": 0.0852982671931386, "clip_ratio/region_mean": 0.14204808231443167, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 59.375, "completions/mean_terminated_length": 59.375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 1.934165135025978, "epoch": 0.011005560704355885, "frac_reward_zero_std": 0.0, "grad_norm": 10.306532859802246, "learning_rate": 9.13939393939394e-06, "loss": 0.0231, "num_tokens": 614381.0, "reward": 0.47206586599349976, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.47206586599349976, "reward_meter_std": 0.39846837520599365, "reward_std": 0.39846834540367126, "reward_total_composite_mean": 0.47206586599349976, "reward_total_composite_std": 0.39846837520599365, "reward_total_mean": 0.47206586599349976, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.47206586599349976, "rewards/meter/std": 0.39846837520599365, "rewards/total_composite/mean": 0.47206586599349976, "rewards/total_composite/std": 0.39846837520599365, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0313324928283691, "sampling/importance_sampling_ratio/min": 0.17719048261642456, "sampling/sampling_logp_difference/max": 1.7305299043655396, "sampling/sampling_logp_difference/mean": 0.16083520650863647, "step": 285 }, { "clip_ratio/high_max": 0.07367282547056675, "clip_ratio/high_mean": 0.07367282547056675, "clip_ratio/low_mean": 0.0790464598685503, "clip_ratio/low_min": 0.0790464598685503, "clip_ratio/region_mean": 0.15271928533911705, "completions/clipped_ratio": 0.0, "completions/max_length": 225.0, "completions/max_terminated_length": 225.0, "completions/mean_length": 177.25, "completions/mean_terminated_length": 177.25, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.8629858121275902, "epoch": 0.01104417670682731, "frac_reward_zero_std": 0.0, "grad_norm": 8.738157272338867, "learning_rate": 9.136363636363637e-06, "loss": 0.1343, "num_tokens": 617527.0, "reward": 0.3599608540534973, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.05050762742757797, "reward_meter_mean": 0.4146118462085724, "reward_meter_std": 0.2753751873970032, "reward_std": 0.23337014019489288, "reward_total_composite_mean": 0.3599608540534973, "reward_total_composite_std": 0.2333701252937317, "reward_total_mean": 0.3599608540534973, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.05050762742757797, "rewards/meter/mean": 0.4146118462085724, "rewards/meter/std": 0.2753751873970032, "rewards/total_composite/mean": 0.3599608540534973, "rewards/total_composite/std": 0.2333701252937317, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9998421669006348, "sampling/importance_sampling_ratio/min": 0.0186123289167881, "sampling/sampling_logp_difference/max": 3.983931064605713, "sampling/sampling_logp_difference/mean": 0.17458735406398773, "step": 286 }, { "clip_ratio/high_max": 0.03732517547905445, "clip_ratio/high_mean": 0.03732517547905445, "clip_ratio/low_mean": 0.1001923680305481, "clip_ratio/low_min": 0.1001923680305481, "clip_ratio/region_mean": 0.13751754350960255, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 129.5, "completions/mean_terminated_length": 129.5, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 1.8472711890935898, "epoch": 0.011082792709298734, "frac_reward_zero_std": 0.0, "grad_norm": 7.0826334953308105, "learning_rate": 9.133333333333335e-06, "loss": -0.003, "num_tokens": 619875.0, "reward": 0.30439293384552, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.38049113750457764, "reward_meter_std": 0.30704429745674133, "reward_std": 0.24563542008399963, "reward_total_composite_mean": 0.30439293384552, "reward_total_composite_std": 0.24563542008399963, "reward_total_mean": 0.30439293384552, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.38049113750457764, "rewards/meter/std": 0.30704429745674133, "rewards/total_composite/mean": 0.30439293384552, "rewards/total_composite/std": 0.24563542008399963, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.029800295829773, "sampling/importance_sampling_ratio/min": 0.230813667178154, "sampling/sampling_logp_difference/max": 1.4661445617675781, "sampling/sampling_logp_difference/mean": 0.1677759736776352, "step": 287 }, { "clip_ratio/high_max": 0.07457270473241806, "clip_ratio/high_mean": 0.07457270473241806, "clip_ratio/low_mean": 0.0725222835317254, "clip_ratio/low_min": 0.0725222835317254, "clip_ratio/region_mean": 0.14709498826414347, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 1.9991831853985786, "epoch": 0.011121408711770158, "frac_reward_zero_std": 0.0, "grad_norm": 9.476400375366211, "learning_rate": 9.130303030303032e-06, "loss": -0.0746, "num_tokens": 621483.0, "reward": 0.6926716566085815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6926716566085815, "reward_meter_std": 0.32063576579093933, "reward_std": 0.32063573598861694, "reward_total_composite_mean": 0.6926716566085815, "reward_total_composite_std": 0.32063576579093933, "reward_total_mean": 0.6926716566085815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6926716566085815, "rewards/meter/std": 0.32063576579093933, "rewards/total_composite/mean": 0.6926716566085815, "rewards/total_composite/std": 0.32063576579093933, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.02389657497406, "sampling/importance_sampling_ratio/min": 0.31524786353111267, "sampling/sampling_logp_difference/max": 1.1543960571289062, "sampling/sampling_logp_difference/mean": 0.1627330631017685, "step": 288 }, { "clip_ratio/high_max": 0.08544671628624201, "clip_ratio/high_mean": 0.08544671628624201, "clip_ratio/low_mean": 0.03081388957798481, "clip_ratio/low_min": 0.03081388957798481, "clip_ratio/region_mean": 0.11626060586422682, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 56.5, "completions/mean_terminated_length": 56.5, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 1.427998036146164, "epoch": 0.011160024714241582, "frac_reward_zero_std": 0.0, "grad_norm": 10.661824226379395, "learning_rate": 9.127272727272727e-06, "loss": 0.093, "num_tokens": 623327.0, "reward": 0.6810826063156128, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6810826063156128, "reward_meter_std": 0.39801594614982605, "reward_std": 0.39801594614982605, "reward_total_composite_mean": 0.6810826063156128, "reward_total_composite_std": 0.39801594614982605, "reward_total_mean": 0.6810826063156128, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6810826063156128, "rewards/meter/std": 0.39801594614982605, "rewards/total_composite/mean": 0.6810826063156128, "rewards/total_composite/std": 0.39801594614982605, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0323865413665771, "sampling/importance_sampling_ratio/min": 0.2761779725551605, "sampling/sampling_logp_difference/max": 1.2867097854614258, "sampling/sampling_logp_difference/mean": 0.12699994444847107, "step": 289 }, { "clip_ratio/high_max": 0.08129793964326382, "clip_ratio/high_mean": 0.08129793964326382, "clip_ratio/low_mean": 0.036112773232162, "clip_ratio/low_min": 0.036112773232162, "clip_ratio/region_mean": 0.11741071287542582, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 91.0, "completions/mean_terminated_length": 91.0, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 1.5821717530488968, "epoch": 0.011198640716713006, "frac_reward_zero_std": 0.0, "grad_norm": 8.565241813659668, "learning_rate": 9.124242424242425e-06, "loss": -0.0217, "num_tokens": 625439.0, "reward": 0.7161059379577637, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7161059379577637, "reward_meter_std": 0.37380164861679077, "reward_std": 0.37380164861679077, "reward_total_composite_mean": 0.7161059379577637, "reward_total_composite_std": 0.37380164861679077, "reward_total_mean": 0.7161059379577637, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7161059379577637, "rewards/meter/std": 0.37380164861679077, "rewards/total_composite/mean": 0.7161059379577637, "rewards/total_composite/std": 0.37380164861679077, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0219635963439941, "sampling/importance_sampling_ratio/min": 0.24008767306804657, "sampling/sampling_logp_difference/max": 1.4267511367797852, "sampling/sampling_logp_difference/mean": 0.1450376659631729, "step": 290 }, { "clip_ratio/high_max": 0.15774553548544645, "clip_ratio/high_mean": 0.15774553548544645, "clip_ratio/low_mean": 0.019999999552965164, "clip_ratio/low_min": 0.019999999552965164, "clip_ratio/region_mean": 0.17774553503841162, "completions/clipped_ratio": 0.0, "completions/max_length": 100.0, "completions/max_terminated_length": 100.0, "completions/mean_length": 84.875, "completions/mean_terminated_length": 84.875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 2.2144875079393387, "epoch": 0.01123725671918443, "frac_reward_zero_std": 0.0, "grad_norm": 9.695059776306152, "learning_rate": 9.121212121212122e-06, "loss": 0.0824, "num_tokens": 627518.0, "reward": 0.8615293502807617, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8615293502807617, "reward_meter_std": 0.3134516179561615, "reward_std": 0.3134515881538391, "reward_total_composite_mean": 0.8615293502807617, "reward_total_composite_std": 0.3134516179561615, "reward_total_mean": 0.8615293502807617, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8615293502807617, "rewards/meter/std": 0.3134516179561615, "rewards/total_composite/mean": 0.8615293502807617, "rewards/total_composite/std": 0.3134516179561615, "sampling/importance_sampling_ratio/max": 1.9597175121307373, "sampling/importance_sampling_ratio/mean": 1.020525574684143, "sampling/importance_sampling_ratio/min": 0.27043434977531433, "sampling/sampling_logp_difference/max": 1.3077259063720703, "sampling/sampling_logp_difference/mean": 0.17841492593288422, "step": 291 }, { "clip_ratio/high_max": 0.050237943418323994, "clip_ratio/high_mean": 0.050237943418323994, "clip_ratio/low_mean": 0.04759177938103676, "clip_ratio/low_min": 0.04759177938103676, "clip_ratio/region_mean": 0.09782972279936075, "completions/clipped_ratio": 0.0, "completions/max_length": 222.0, "completions/max_terminated_length": 222.0, "completions/mean_length": 199.25, "completions/mean_terminated_length": 199.25, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 1.7579984441399574, "epoch": 0.011275872721655854, "frac_reward_zero_std": 0.0, "grad_norm": 4.9265546798706055, "learning_rate": 9.118181818181819e-06, "loss": 0.0485, "num_tokens": 630696.0, "reward": 0.4524690806865692, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5907192230224609, "reward_meter_std": 0.3661164939403534, "reward_std": 0.37379953265190125, "reward_total_composite_mean": 0.4524690806865692, "reward_total_composite_std": 0.37379953265190125, "reward_total_mean": 0.4524690806865692, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5907192230224609, "rewards/meter/std": 0.3661164939403534, "rewards/total_composite/mean": 0.4524690806865692, "rewards/total_composite/std": 0.37379953265190125, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0290663242340088, "sampling/importance_sampling_ratio/min": 0.2462177872657776, "sampling/sampling_logp_difference/max": 1.4015388488769531, "sampling/sampling_logp_difference/mean": 0.14518797397613525, "step": 292 }, { "clip_ratio/high_max": 0.10475165769457817, "clip_ratio/high_mean": 0.10475165769457817, "clip_ratio/low_mean": 0.051651342771947384, "clip_ratio/low_min": 0.051651342771947384, "clip_ratio/region_mean": 0.15640300046652555, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 64.25, "completions/mean_terminated_length": 64.25, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 1.85820072889328, "epoch": 0.011314488724127278, "frac_reward_zero_std": 0.0, "grad_norm": 9.675642967224121, "learning_rate": 9.115151515151516e-06, "loss": 0.0185, "num_tokens": 632410.0, "reward": 0.9181022047996521, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9181022047996521, "reward_meter_std": 0.09518402069807053, "reward_std": 0.09518400579690933, "reward_total_composite_mean": 0.9181022047996521, "reward_total_composite_std": 0.09518402069807053, "reward_total_mean": 0.9181022047996521, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9181022047996521, "rewards/meter/std": 0.09518402069807053, "rewards/total_composite/mean": 0.9181022047996521, "rewards/total_composite/std": 0.09518402069807053, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.036954402923584, "sampling/importance_sampling_ratio/min": 0.16399513185024261, "sampling/sampling_logp_difference/max": 1.8079185485839844, "sampling/sampling_logp_difference/mean": 0.15963061153888702, "step": 293 }, { "clip_ratio/high_max": 0.15911428444087505, "clip_ratio/high_mean": 0.15911428444087505, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/region_mean": 0.16804285626858473, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 31.0, "completions/mean_terminated_length": 31.0, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 2.2211313992738724, "epoch": 0.011353104726598702, "frac_reward_zero_std": 0.0, "grad_norm": 22.611948013305664, "learning_rate": 9.112121212121214e-06, "loss": -0.022, "num_tokens": 633858.0, "reward": 0.8932417631149292, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8932417631149292, "reward_meter_std": 0.2514517903327942, "reward_std": 0.2514517605304718, "reward_total_composite_mean": 0.8932417631149292, "reward_total_composite_std": 0.2514517903327942, "reward_total_mean": 0.8932417631149292, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8932417631149292, "rewards/meter/std": 0.2514517903327942, "rewards/total_composite/mean": 0.8932417631149292, "rewards/total_composite/std": 0.2514517903327942, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0493816137313843, "sampling/importance_sampling_ratio/min": 0.11040541529655457, "sampling/sampling_logp_difference/max": 2.2035961151123047, "sampling/sampling_logp_difference/mean": 0.18085245788097382, "step": 294 }, { "clip_ratio/high_max": 0.08510276302695274, "clip_ratio/high_mean": 0.08510276302695274, "clip_ratio/low_mean": 0.05079313600435853, "clip_ratio/low_min": 0.05079313600435853, "clip_ratio/region_mean": 0.13589589903131127, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 63.375, "completions/mean_terminated_length": 63.375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 1.5011917799711227, "epoch": 0.011391720729070126, "frac_reward_zero_std": 0.0, "grad_norm": 11.888802528381348, "learning_rate": 9.10909090909091e-06, "loss": 0.05, "num_tokens": 635653.0, "reward": 0.7532914876937866, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7532914876937866, "reward_meter_std": 0.3347664475440979, "reward_std": 0.3347664773464203, "reward_total_composite_mean": 0.7532914876937866, "reward_total_composite_std": 0.3347664475440979, "reward_total_mean": 0.7532914876937866, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7532914876937866, "rewards/meter/std": 0.3347664475440979, "rewards/total_composite/mean": 0.7532914876937866, "rewards/total_composite/std": 0.3347664475440979, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0173488855361938, "sampling/importance_sampling_ratio/min": 0.20713002979755402, "sampling/sampling_logp_difference/max": 1.5744085311889648, "sampling/sampling_logp_difference/mean": 0.1385606974363327, "step": 295 }, { "clip_ratio/high_max": 0.054404761642217636, "clip_ratio/high_mean": 0.054404761642217636, "clip_ratio/low_mean": 0.06392320711165667, "clip_ratio/low_min": 0.06392320711165667, "clip_ratio/region_mean": 0.1183279687538743, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 25.875, "completions/mean_terminated_length": 25.875, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.7283558156341314, "epoch": 0.01143033673154155, "frac_reward_zero_std": 0.0, "grad_norm": 37.00153732299805, "learning_rate": 9.106060606060606e-06, "loss": 0.1775, "num_tokens": 637188.0, "reward": 0.8021076321601868, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8021076321601868, "reward_meter_std": 0.21438796818256378, "reward_std": 0.21438796818256378, "reward_total_composite_mean": 0.8021076321601868, "reward_total_composite_std": 0.21438796818256378, "reward_total_mean": 0.8021076321601868, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8021076321601868, "rewards/meter/std": 0.21438796818256378, "rewards/total_composite/mean": 0.8021076321601868, "rewards/total_composite/std": 0.21438796818256378, "sampling/importance_sampling_ratio/max": 1.9617925882339478, "sampling/importance_sampling_ratio/mean": 0.9893893003463745, "sampling/importance_sampling_ratio/min": 0.07945267111063004, "sampling/sampling_logp_difference/max": 2.5325937271118164, "sampling/sampling_logp_difference/mean": 0.15415281057357788, "step": 296 }, { "clip_ratio/high_max": 0.06547143869102001, "clip_ratio/high_mean": 0.06547143869102001, "clip_ratio/low_mean": 0.046736557967960835, "clip_ratio/low_min": 0.046736557967960835, "clip_ratio/region_mean": 0.11220799665898085, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 90.125, "completions/mean_terminated_length": 90.125, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 1.3437079191207886, "epoch": 0.011468952734012975, "frac_reward_zero_std": 0.0, "grad_norm": 10.021339416503906, "learning_rate": 9.103030303030304e-06, "loss": 0.0003, "num_tokens": 639349.0, "reward": 0.6162939667701721, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6162939667701721, "reward_meter_std": 0.46589404344558716, "reward_std": 0.46589401364326477, "reward_total_composite_mean": 0.6162939667701721, "reward_total_composite_std": 0.46589404344558716, "reward_total_mean": 0.6162939667701721, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6162939667701721, "rewards/meter/std": 0.46589404344558716, "rewards/total_composite/mean": 0.6162939667701721, "rewards/total_composite/std": 0.46589404344558716, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0367764234542847, "sampling/importance_sampling_ratio/min": 0.24083733558654785, "sampling/sampling_logp_difference/max": 1.4236335754394531, "sampling/sampling_logp_difference/mean": 0.13409848511219025, "step": 297 }, { "clip_ratio/high_max": 0.0976903848350048, "clip_ratio/high_mean": 0.0976903848350048, "clip_ratio/low_mean": 0.03513659443706274, "clip_ratio/low_min": 0.03513659443706274, "clip_ratio/region_mean": 0.13282697927206755, "completions/clipped_ratio": 0.0, "completions/max_length": 212.0, "completions/max_terminated_length": 212.0, "completions/mean_length": 183.25, "completions/mean_terminated_length": 183.25, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 1.7826667726039886, "epoch": 0.011507568736484399, "frac_reward_zero_std": 0.0, "grad_norm": 6.047693729400635, "learning_rate": 9.100000000000001e-06, "loss": 0.0128, "num_tokens": 642375.0, "reward": 0.7657866477966309, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8214285373687744, "reward_count_adherence_std": 0.06613000482320786, "reward_meter_mean": 0.9324739575386047, "reward_meter_std": 0.09105537831783295, "reward_std": 0.09787716716527939, "reward_total_composite_mean": 0.7657866477966309, "reward_total_composite_std": 0.09787718951702118, "reward_total_mean": 0.7657866477966309, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8214285373687744, "rewards/count_adherence/std": 0.06613000482320786, "rewards/meter/mean": 0.9324739575386047, "rewards/meter/std": 0.09105537831783295, "rewards/total_composite/mean": 0.7657866477966309, "rewards/total_composite/std": 0.09787718951702118, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0216869115829468, "sampling/importance_sampling_ratio/min": 0.02991897612810135, "sampling/sampling_logp_difference/max": 3.5092623233795166, "sampling/sampling_logp_difference/mean": 0.14809899032115936, "step": 298 }, { "clip_ratio/high_max": 0.10818896908313036, "clip_ratio/high_mean": 0.10818896908313036, "clip_ratio/low_mean": 0.046733343973755836, "clip_ratio/low_min": 0.046733343973755836, "clip_ratio/region_mean": 0.1549223130568862, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 65.25, "completions/mean_terminated_length": 65.25, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 1.74993297457695, "epoch": 0.011546184738955823, "frac_reward_zero_std": 0.0, "grad_norm": 12.910846710205078, "learning_rate": 9.096969696969698e-06, "loss": 0.1033, "num_tokens": 644161.0, "reward": 0.7445110082626343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.8054907917976379, "reward_meter_std": 0.36134690046310425, "reward_std": 0.3695930242538452, "reward_total_composite_mean": 0.7445110082626343, "reward_total_composite_std": 0.3695930540561676, "reward_total_mean": 0.7445110082626343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.8054907917976379, "rewards/meter/std": 0.36134690046310425, "rewards/total_composite/mean": 0.7445110082626343, "rewards/total_composite/std": 0.3695930540561676, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0238062143325806, "sampling/importance_sampling_ratio/min": 0.26225021481513977, "sampling/sampling_logp_difference/max": 1.338456153869629, "sampling/sampling_logp_difference/mean": 0.16549576818943024, "step": 299 }, { "clip_ratio/high_max": 0.09760686475783587, "clip_ratio/high_mean": 0.09760686475783587, "clip_ratio/low_mean": 0.06600985117256641, "clip_ratio/low_min": 0.06600985117256641, "clip_ratio/region_mean": 0.16361671593040228, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 28.875, "completions/mean_terminated_length": 28.875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 2.053573176264763, "epoch": 0.011584800741427247, "frac_reward_zero_std": 0.0, "grad_norm": 13.964972496032715, "learning_rate": 9.093939393939395e-06, "loss": 0.0899, "num_tokens": 645560.0, "reward": 0.6676859259605408, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6676859259605408, "reward_meter_std": 0.40630266070365906, "reward_std": 0.40630266070365906, "reward_total_composite_mean": 0.6676859259605408, "reward_total_composite_std": 0.40630266070365906, "reward_total_mean": 0.6676859259605408, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6676859259605408, "rewards/meter/std": 0.40630266070365906, "rewards/total_composite/mean": 0.6676859259605408, "rewards/total_composite/std": 0.40630266070365906, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0266624689102173, "sampling/importance_sampling_ratio/min": 0.29776731133461, "sampling/sampling_logp_difference/max": 1.2114429473876953, "sampling/sampling_logp_difference/mean": 0.17699161171913147, "step": 300 }, { "epoch": 0.011584800741427247, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/max_length": 413.84615384615387, "eval_completions/max_terminated_length": 332.6923076923077, "eval_completions/mean_length": 189.27884615384616, "eval_completions/mean_terminated_length": 169.91621281550482, "eval_completions/min_length": 48.69230769230769, "eval_completions/min_terminated_length": 48.69230769230769, "eval_entropy": 1.6832339671941905, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 645560.0, "eval_reward": 0.47910642050779784, "eval_reward_arabic_clean_mean": 0.9326923076923077, "eval_reward_arabic_clean_std": 0.19037489936901972, "eval_reward_count_adherence_mean": 0.8583941322106582, "eval_reward_count_adherence_std": 0.16477382297699267, "eval_reward_meter_mean": 0.5671234520582052, "eval_reward_meter_std": 0.3898524023019351, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.47910642050779784, "eval_reward_total_composite_std": 0.3604818215736976, "eval_reward_total_mean": 0.47910642050779784, "eval_rewards/arabic_clean/mean": 0.9326923076923077, "eval_rewards/arabic_clean/std": 0.19037489936901972, "eval_rewards/count_adherence/mean": 0.8583941322106582, "eval_rewards/count_adherence/std": 0.16477382297699267, "eval_rewards/meter/mean": 0.5671234520582052, "eval_rewards/meter/std": 0.3898524023019351, "eval_rewards/total_composite/mean": 0.47910642050779784, "eval_rewards/total_composite/std": 0.3604818215736976, "eval_runtime": 76.9936, "eval_samples_per_second": 1.351, "eval_sampling/importance_sampling_ratio/max": 1.5635485924207246, "eval_sampling/importance_sampling_ratio/mean": 1.0292121997246375, "eval_sampling/importance_sampling_ratio/min": 0.32239050360826343, "eval_sampling/sampling_logp_difference/max": 1.1390617810762846, "eval_sampling/sampling_logp_difference/mean": 0.10725439225251858, "eval_steps_per_second": 0.169, "step": 300 }, { "clip_ratio/high_max": 0.021113280206918716, "clip_ratio/high_mean": 0.021113280206918716, "clip_ratio/low_mean": 0.012804877944290638, "clip_ratio/low_min": 0.012804877944290638, "clip_ratio/region_mean": 0.033918158151209354, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 462.625, "completions/mean_terminated_length": 380.3333435058594, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "entropy": 0.7612107247114182, "epoch": 0.011623416743898671, "frac_reward_zero_std": 0.0, "grad_norm": 1.71794855594635, "learning_rate": 9.090909090909091e-06, "loss": -0.313, "num_tokens": 648373.0, "reward": 0.21318131685256958, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_count_adherence_mean": 0.4375, "reward_count_adherence_std": 0.18169663846492767, "reward_meter_mean": 0.6754355430603027, "reward_meter_std": 0.3314078450202942, "reward_std": 0.2553112506866455, "reward_total_composite_mean": 0.21318131685256958, "reward_total_composite_std": 0.2553112804889679, "reward_total_mean": 0.21318131685256958, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/count_adherence/mean": 0.4375, "rewards/count_adherence/std": 0.18169663846492767, "rewards/meter/mean": 0.6754355430603027, "rewards/meter/std": 0.3314078450202942, "rewards/total_composite/mean": 0.21318131685256958, "rewards/total_composite/std": 0.2553112804889679, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0243096351623535, "sampling/importance_sampling_ratio/min": 0.2710690200328827, "sampling/sampling_logp_difference/max": 1.3053817749023438, "sampling/sampling_logp_difference/mean": 0.13692021369934082, "step": 301 }, { "clip_ratio/high_max": 0.10167329479008913, "clip_ratio/high_mean": 0.10167329479008913, "clip_ratio/low_mean": 0.06732289679348469, "clip_ratio/low_min": 0.06732289679348469, "clip_ratio/region_mean": 0.16899619158357382, "completions/clipped_ratio": 0.0, "completions/max_length": 156.0, "completions/max_terminated_length": 156.0, "completions/mean_length": 131.875, "completions/mean_terminated_length": 131.875, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 1.9741446822881699, "epoch": 0.011662032746370095, "frac_reward_zero_std": 0.0, "grad_norm": 6.904706954956055, "learning_rate": 9.087878787878788e-06, "loss": -0.0483, "num_tokens": 651028.0, "reward": 0.3257467448711395, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.407183438539505, "reward_meter_std": 0.17483185231685638, "reward_std": 0.1398654729127884, "reward_total_composite_mean": 0.3257467448711395, "reward_total_composite_std": 0.1398654729127884, "reward_total_mean": 0.3257467448711395, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.407183438539505, "rewards/meter/std": 0.17483185231685638, "rewards/total_composite/mean": 0.3257467448711395, "rewards/total_composite/std": 0.1398654729127884, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.031490445137024, "sampling/importance_sampling_ratio/min": 0.20558007061481476, "sampling/sampling_logp_difference/max": 2.081483840942383, "sampling/sampling_logp_difference/mean": 0.17297177016735077, "step": 302 }, { "clip_ratio/high_max": 0.0840634061023593, "clip_ratio/high_mean": 0.0840634061023593, "clip_ratio/low_mean": 0.03281614277511835, "clip_ratio/low_min": 0.03281614277511835, "clip_ratio/region_mean": 0.11687954887747765, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 35.375, "completions/mean_terminated_length": 35.375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 1.5965645760297775, "epoch": 0.01170064874884152, "frac_reward_zero_std": 0.0, "grad_norm": 13.510102272033691, "learning_rate": 9.084848484848486e-06, "loss": -0.0003, "num_tokens": 652679.0, "reward": 0.6417617797851562, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6417617797851562, "reward_meter_std": 0.47610655426979065, "reward_std": 0.47610652446746826, "reward_total_composite_mean": 0.6417617797851562, "reward_total_composite_std": 0.47610655426979065, "reward_total_mean": 0.6417617797851562, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6417617797851562, "rewards/meter/std": 0.47610655426979065, "rewards/total_composite/mean": 0.6417617797851562, "rewards/total_composite/std": 0.47610655426979065, "sampling/importance_sampling_ratio/max": 1.849202036857605, "sampling/importance_sampling_ratio/mean": 1.001904010772705, "sampling/importance_sampling_ratio/min": 0.16444538533687592, "sampling/sampling_logp_difference/max": 1.8051767349243164, "sampling/sampling_logp_difference/mean": 0.16745901107788086, "step": 303 }, { "clip_ratio/high_max": 0.08884815126657486, "clip_ratio/high_mean": 0.08884815126657486, "clip_ratio/low_mean": 0.033333334140479565, "clip_ratio/low_min": 0.033333334140479565, "clip_ratio/region_mean": 0.12218148540705442, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 65.5, "completions/mean_terminated_length": 65.5, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 1.989267036318779, "epoch": 0.011739264751312943, "frac_reward_zero_std": 0.0, "grad_norm": 9.380762100219727, "learning_rate": 9.081818181818183e-06, "loss": -0.0536, "num_tokens": 654531.0, "reward": 0.5989729166030884, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6257134079933167, "reward_meter_std": 0.39615458250045776, "reward_std": 0.43339118361473083, "reward_total_composite_mean": 0.5989729166030884, "reward_total_composite_std": 0.43339118361473083, "reward_total_mean": 0.5989729166030884, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6257134079933167, "rewards/meter/std": 0.39615458250045776, "rewards/total_composite/mean": 0.5989729166030884, "rewards/total_composite/std": 0.43339118361473083, "sampling/importance_sampling_ratio/max": 1.9768328666687012, "sampling/importance_sampling_ratio/mean": 1.0195375680923462, "sampling/importance_sampling_ratio/min": 0.1475263237953186, "sampling/sampling_logp_difference/max": 1.9137487411499023, "sampling/sampling_logp_difference/mean": 0.174082413315773, "step": 304 }, { "clip_ratio/high_max": 0.10270242113620043, "clip_ratio/high_mean": 0.10270242113620043, "clip_ratio/low_mean": 0.027922078035771847, "clip_ratio/low_min": 0.027922078035771847, "clip_ratio/region_mean": 0.13062449917197227, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 1.5056721195578575, "epoch": 0.011777880753784368, "frac_reward_zero_std": 0.0, "grad_norm": 10.915406227111816, "learning_rate": 9.078787878787878e-06, "loss": 0.0498, "num_tokens": 656283.0, "reward": 0.8347302079200745, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8347302079200745, "reward_meter_std": 0.3262871503829956, "reward_std": 0.3262871503829956, "reward_total_composite_mean": 0.8347302079200745, "reward_total_composite_std": 0.3262871503829956, "reward_total_mean": 0.8347302079200745, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8347302079200745, "rewards/meter/std": 0.3262871503829956, "rewards/total_composite/mean": 0.8347302079200745, "rewards/total_composite/std": 0.3262871503829956, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0057920217514038, "sampling/importance_sampling_ratio/min": 0.21743382513523102, "sampling/sampling_logp_difference/max": 1.5258607864379883, "sampling/sampling_logp_difference/mean": 0.1379973143339157, "step": 305 }, { "clip_ratio/high_max": 0.10306308604776859, "clip_ratio/high_mean": 0.10306308604776859, "clip_ratio/low_mean": 0.09684343822300434, "clip_ratio/low_min": 0.09684343822300434, "clip_ratio/region_mean": 0.19990652427077293, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 21.75, "completions/mean_terminated_length": 21.75, "completions/min_length": 15.0, "completions/min_terminated_length": 15.0, "entropy": 2.3475464656949043, "epoch": 0.011816496756255792, "frac_reward_zero_std": 0.0, "grad_norm": 15.107892036437988, "learning_rate": 9.075757575757577e-06, "loss": -0.0637, "num_tokens": 657681.0, "reward": 0.6698324680328369, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6698324680328369, "reward_meter_std": 0.42919304966926575, "reward_std": 0.42919301986694336, "reward_total_composite_mean": 0.6698324680328369, "reward_total_composite_std": 0.42919304966926575, "reward_total_mean": 0.6698324680328369, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6698324680328369, "rewards/meter/std": 0.42919304966926575, "rewards/total_composite/mean": 0.6698324680328369, "rewards/total_composite/std": 0.42919304966926575, "sampling/importance_sampling_ratio/max": 1.7879084348678589, "sampling/importance_sampling_ratio/mean": 1.038679838180542, "sampling/importance_sampling_ratio/min": 0.3113321363925934, "sampling/sampling_logp_difference/max": 1.1668949127197266, "sampling/sampling_logp_difference/mean": 0.18782852590084076, "step": 306 }, { "clip_ratio/high_max": 0.064244344830513, "clip_ratio/high_mean": 0.064244344830513, "clip_ratio/low_mean": 0.08673457149416208, "clip_ratio/low_min": 0.08673457149416208, "clip_ratio/region_mean": 0.15097891632467508, "completions/clipped_ratio": 0.0, "completions/max_length": 131.0, "completions/max_terminated_length": 131.0, "completions/mean_length": 103.5, "completions/mean_terminated_length": 103.5, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 2.192972496151924, "epoch": 0.011855112758727216, "frac_reward_zero_std": 0.0, "grad_norm": 7.24405574798584, "learning_rate": 9.072727272727273e-06, "loss": 0.0343, "num_tokens": 659813.0, "reward": 0.4702509343624115, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5797778367996216, "reward_meter_std": 0.4139650762081146, "reward_std": 0.4394586980342865, "reward_total_composite_mean": 0.4702509343624115, "reward_total_composite_std": 0.4394587278366089, "reward_total_mean": 0.4702509343624115, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5797778367996216, "rewards/meter/std": 0.4139650762081146, "rewards/total_composite/mean": 0.4702509343624115, "rewards/total_composite/std": 0.4394587278366089, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0307226181030273, "sampling/importance_sampling_ratio/min": 0.19179849326610565, "sampling/sampling_logp_difference/max": 1.6513099670410156, "sampling/sampling_logp_difference/mean": 0.16522862017154694, "step": 307 }, { "clip_ratio/high_max": 0.05581671930849552, "clip_ratio/high_mean": 0.05581671930849552, "clip_ratio/low_mean": 0.07120703207328916, "clip_ratio/low_min": 0.07120703207328916, "clip_ratio/region_mean": 0.12702375138178468, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 86.5, "completions/mean_terminated_length": 86.5, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 1.4088008925318718, "epoch": 0.011893728761198642, "frac_reward_zero_std": 0.0, "grad_norm": 12.200899124145508, "learning_rate": 9.06969696969697e-06, "loss": -0.0505, "num_tokens": 661897.0, "reward": 0.5569584369659424, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.5630761384963989, "reward_meter_std": 0.37055954337120056, "reward_std": 0.3778226375579834, "reward_total_composite_mean": 0.5569584369659424, "reward_total_composite_std": 0.3778226375579834, "reward_total_mean": 0.5569584369659424, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.5630761384963989, "rewards/meter/std": 0.37055954337120056, "rewards/total_composite/mean": 0.5569584369659424, "rewards/total_composite/std": 0.3778226375579834, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0206682682037354, "sampling/importance_sampling_ratio/min": 0.13020169734954834, "sampling/sampling_logp_difference/max": 2.038670539855957, "sampling/sampling_logp_difference/mean": 0.16500020027160645, "step": 308 }, { "clip_ratio/high_max": 0.08292348962277174, "clip_ratio/high_mean": 0.08292348962277174, "clip_ratio/low_mean": 0.06721637584269047, "clip_ratio/low_min": 0.06721637584269047, "clip_ratio/region_mean": 0.1501398654654622, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 84.875, "completions/mean_terminated_length": 84.875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 2.261012151837349, "epoch": 0.011932344763670066, "frac_reward_zero_std": 0.0, "grad_norm": 9.461854934692383, "learning_rate": 9.066666666666667e-06, "loss": -0.0083, "num_tokens": 663904.0, "reward": 0.607955813407898, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.6280455589294434, "reward_meter_std": 0.17267122864723206, "reward_std": 0.19935740530490875, "reward_total_composite_mean": 0.607955813407898, "reward_total_composite_std": 0.19935742020606995, "reward_total_mean": 0.607955813407898, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.6280455589294434, "rewards/meter/std": 0.17267122864723206, "rewards/total_composite/mean": 0.607955813407898, "rewards/total_composite/std": 0.19935742020606995, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0375332832336426, "sampling/importance_sampling_ratio/min": 0.2669237554073334, "sampling/sampling_logp_difference/max": 1.3207921981811523, "sampling/sampling_logp_difference/mean": 0.17172668874263763, "step": 309 }, { "clip_ratio/high_max": 0.06727708876132965, "clip_ratio/high_mean": 0.06727708876132965, "clip_ratio/low_mean": 0.061003027483820915, "clip_ratio/low_min": 0.061003027483820915, "clip_ratio/region_mean": 0.12828011624515057, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 129.125, "completions/mean_terminated_length": 129.125, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 1.74216927587986, "epoch": 0.01197096076614149, "frac_reward_zero_std": 0.0, "grad_norm": 6.23518705368042, "learning_rate": 9.063636363636365e-06, "loss": -0.0331, "num_tokens": 666273.0, "reward": 0.6215674877166748, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.7603760957717896, "reward_meter_std": 0.25764790177345276, "reward_std": 0.19575539231300354, "reward_total_composite_mean": 0.6215674877166748, "reward_total_composite_std": 0.19575539231300354, "reward_total_mean": 0.6215674877166748, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.7603760957717896, "rewards/meter/std": 0.25764790177345276, "rewards/total_composite/mean": 0.6215674877166748, "rewards/total_composite/std": 0.19575539231300354, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0316812992095947, "sampling/importance_sampling_ratio/min": 0.21927233040332794, "sampling/sampling_logp_difference/max": 1.5174407958984375, "sampling/sampling_logp_difference/mean": 0.14828266203403473, "step": 310 }, { "clip_ratio/high_max": 0.1306404136121273, "clip_ratio/high_mean": 0.1306404136121273, "clip_ratio/low_mean": 0.06916317716240883, "clip_ratio/low_min": 0.06916317716240883, "clip_ratio/region_mean": 0.19980359077453613, "completions/clipped_ratio": 0.0, "completions/max_length": 203.0, "completions/max_terminated_length": 203.0, "completions/mean_length": 141.625, "completions/mean_terminated_length": 141.625, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 1.3266419470310211, "epoch": 0.012009576768612914, "frac_reward_zero_std": 0.0, "grad_norm": 15.047697067260742, "learning_rate": 9.06060606060606e-06, "loss": 0.1673, "num_tokens": 668998.0, "reward": 0.7146421670913696, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.953125, "reward_count_adherence_std": 0.06469365209341049, "reward_meter_mean": 0.7493094205856323, "reward_meter_std": 0.22875480353832245, "reward_std": 0.2316495180130005, "reward_total_composite_mean": 0.7146421670913696, "reward_total_composite_std": 0.2316495180130005, "reward_total_mean": 0.7146421670913696, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.953125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/meter/mean": 0.7493094205856323, "rewards/meter/std": 0.22875480353832245, "rewards/total_composite/mean": 0.7146421670913696, "rewards/total_composite/std": 0.2316495180130005, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9966016411781311, "sampling/importance_sampling_ratio/min": 0.041286900639534, "sampling/sampling_logp_difference/max": 3.1872100830078125, "sampling/sampling_logp_difference/mean": 0.21245455741882324, "step": 311 }, { "clip_ratio/high_max": 0.15273912716656923, "clip_ratio/high_mean": 0.15273912716656923, "clip_ratio/low_mean": 0.030448718927800655, "clip_ratio/low_min": 0.030448718927800655, "clip_ratio/region_mean": 0.1831878460943699, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 30.375, "completions/mean_terminated_length": 30.375, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 1.9258029609918594, "epoch": 0.012048192771084338, "frac_reward_zero_std": 0.0, "grad_norm": 19.782384872436523, "learning_rate": 9.057575757575759e-06, "loss": -0.011, "num_tokens": 670353.0, "reward": 0.7651474475860596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7651474475860596, "reward_meter_std": 0.3580241799354553, "reward_std": 0.35802415013313293, "reward_total_composite_mean": 0.7651474475860596, "reward_total_composite_std": 0.3580241799354553, "reward_total_mean": 0.7651474475860596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7651474475860596, "rewards/meter/std": 0.3580241799354553, "rewards/total_composite/mean": 0.7651474475860596, "rewards/total_composite/std": 0.3580241799354553, "sampling/importance_sampling_ratio/max": 1.8821232318878174, "sampling/importance_sampling_ratio/mean": 1.0236084461212158, "sampling/importance_sampling_ratio/min": 0.33582067489624023, "sampling/sampling_logp_difference/max": 1.0911779403686523, "sampling/sampling_logp_difference/mean": 0.16778768599033356, "step": 312 }, { "clip_ratio/high_max": 0.10207792790606618, "clip_ratio/high_mean": 0.10207792790606618, "clip_ratio/low_mean": 0.03583333361893892, "clip_ratio/low_min": 0.03583333361893892, "clip_ratio/region_mean": 0.1379112615250051, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 62.0, "completions/mean_terminated_length": 62.0, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 2.047884911298752, "epoch": 0.012086808773555762, "frac_reward_zero_std": 0.0, "grad_norm": 10.562493324279785, "learning_rate": 9.054545454545455e-06, "loss": -0.0656, "num_tokens": 671977.0, "reward": 0.9314479231834412, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9314479231834412, "reward_meter_std": 0.10455489158630371, "reward_std": 0.10455489158630371, "reward_total_composite_mean": 0.9314479231834412, "reward_total_composite_std": 0.10455489158630371, "reward_total_mean": 0.9314479231834412, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9314479231834412, "rewards/meter/std": 0.10455489158630371, "rewards/total_composite/mean": 0.9314479231834412, "rewards/total_composite/std": 0.10455489158630371, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0549638271331787, "sampling/importance_sampling_ratio/min": 0.23077911138534546, "sampling/sampling_logp_difference/max": 1.466294288635254, "sampling/sampling_logp_difference/mean": 0.1677490621805191, "step": 313 }, { "clip_ratio/high_max": 0.12252288311719894, "clip_ratio/high_mean": 0.12252288311719894, "clip_ratio/low_mean": 0.02873883955180645, "clip_ratio/low_min": 0.02873883955180645, "clip_ratio/region_mean": 0.1512617226690054, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 62.125, "completions/mean_terminated_length": 62.125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 1.8069640547037125, "epoch": 0.012125424776027186, "frac_reward_zero_std": 0.0, "grad_norm": 16.10173797607422, "learning_rate": 9.051515151515152e-06, "loss": -0.0042, "num_tokens": 673690.0, "reward": 0.8384339809417725, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8384339809417725, "reward_meter_std": 0.33158746361732483, "reward_std": 0.33158746361732483, "reward_total_composite_mean": 0.8384339809417725, "reward_total_composite_std": 0.33158746361732483, "reward_total_mean": 0.8384339809417725, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8384339809417725, "rewards/meter/std": 0.33158746361732483, "rewards/total_composite/mean": 0.8384339809417725, "rewards/total_composite/std": 0.33158746361732483, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0274717807769775, "sampling/importance_sampling_ratio/min": 0.19091176986694336, "sampling/sampling_logp_difference/max": 1.6559438705444336, "sampling/sampling_logp_difference/mean": 0.15842776000499725, "step": 314 }, { "clip_ratio/high_max": 0.09765755478292704, "clip_ratio/high_mean": 0.09765755478292704, "clip_ratio/low_mean": 0.016509434208273888, "clip_ratio/low_min": 0.016509434208273888, "clip_ratio/region_mean": 0.11416698899120092, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 111.75, "completions/mean_terminated_length": 111.75, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 1.6218044012784958, "epoch": 0.01216404077849861, "frac_reward_zero_std": 0.0, "grad_norm": 6.204983234405518, "learning_rate": 9.04848484848485e-06, "loss": -0.01, "num_tokens": 676000.0, "reward": 0.9871163368225098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9871163368225098, "reward_meter_std": 0.022878479212522507, "reward_std": 0.022878482937812805, "reward_total_composite_mean": 0.9871163368225098, "reward_total_composite_std": 0.022878479212522507, "reward_total_mean": 0.9871163368225098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9871163368225098, "rewards/meter/std": 0.022878479212522507, "rewards/total_composite/mean": 0.9871163368225098, "rewards/total_composite/std": 0.022878479212522507, "sampling/importance_sampling_ratio/max": 1.6907645463943481, "sampling/importance_sampling_ratio/mean": 1.0214686393737793, "sampling/importance_sampling_ratio/min": 0.27453866600990295, "sampling/sampling_logp_difference/max": 1.2926632165908813, "sampling/sampling_logp_difference/mean": 0.13204924762248993, "step": 315 }, { "clip_ratio/high_max": 0.0503198578953743, "clip_ratio/high_mean": 0.0503198578953743, "clip_ratio/low_mean": 0.0671336529776454, "clip_ratio/low_min": 0.0671336529776454, "clip_ratio/region_mean": 0.1174535108730197, "completions/clipped_ratio": 0.0, "completions/max_length": 173.0, "completions/max_terminated_length": 173.0, "completions/mean_length": 161.0, "completions/mean_terminated_length": 161.0, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "entropy": 1.385835349559784, "epoch": 0.012202656780970034, "frac_reward_zero_std": 0.0, "grad_norm": 6.30336332321167, "learning_rate": 9.045454545454546e-06, "loss": 0.0004, "num_tokens": 678856.0, "reward": 0.343695729970932, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.412434846162796, "reward_meter_std": 0.3995774984359741, "reward_std": 0.3329812288284302, "reward_total_composite_mean": 0.343695729970932, "reward_total_composite_std": 0.33298125863075256, "reward_total_mean": 0.343695729970932, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.412434846162796, "rewards/meter/std": 0.3995774984359741, "rewards/total_composite/mean": 0.343695729970932, "rewards/total_composite/std": 0.33298125863075256, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0196648836135864, "sampling/importance_sampling_ratio/min": 0.18164780735969543, "sampling/sampling_logp_difference/max": 1.7056856155395508, "sampling/sampling_logp_difference/mean": 0.14012373983860016, "step": 316 }, { "clip_ratio/high_max": 0.13084796536713839, "clip_ratio/high_mean": 0.13084796536713839, "clip_ratio/low_mean": 0.020408162847161293, "clip_ratio/low_min": 0.020408162847161293, "clip_ratio/region_mean": 0.15125612821429968, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 59.125, "completions/mean_terminated_length": 59.125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 1.7790979892015457, "epoch": 0.012241272783441458, "frac_reward_zero_std": 0.0, "grad_norm": 11.422844886779785, "learning_rate": 9.042424242424244e-06, "loss": -0.0392, "num_tokens": 680465.0, "reward": 0.9061346650123596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9061346650123596, "reward_meter_std": 0.22870448231697083, "reward_std": 0.22870448231697083, "reward_total_composite_mean": 0.9061346650123596, "reward_total_composite_std": 0.22870448231697083, "reward_total_mean": 0.9061346650123596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9061346650123596, "rewards/meter/std": 0.22870448231697083, "rewards/total_composite/mean": 0.9061346650123596, "rewards/total_composite/std": 0.22870448231697083, "sampling/importance_sampling_ratio/max": 1.9018646478652954, "sampling/importance_sampling_ratio/mean": 1.0229982137680054, "sampling/importance_sampling_ratio/min": 0.29680660367012024, "sampling/sampling_logp_difference/max": 1.214674472808838, "sampling/sampling_logp_difference/mean": 0.14648805558681488, "step": 317 }, { "clip_ratio/high_max": 0.13356309290975332, "clip_ratio/high_mean": 0.13356309290975332, "clip_ratio/low_mean": 0.015384615398943424, "clip_ratio/low_min": 0.015384615398943424, "clip_ratio/region_mean": 0.14894770830869675, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 70.375, "completions/mean_terminated_length": 70.375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 2.0886534601449966, "epoch": 0.012279888785912883, "frac_reward_zero_std": 0.0, "grad_norm": 8.7025728225708, "learning_rate": 9.03939393939394e-06, "loss": -0.0306, "num_tokens": 682172.0, "reward": 0.9199281930923462, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9199281930923462, "reward_meter_std": 0.17555859684944153, "reward_std": 0.17555858194828033, "reward_total_composite_mean": 0.9199281930923462, "reward_total_composite_std": 0.17555859684944153, "reward_total_mean": 0.9199281930923462, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9199281930923462, "rewards/meter/std": 0.17555859684944153, "rewards/total_composite/mean": 0.9199281930923462, "rewards/total_composite/std": 0.17555859684944153, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0345814228057861, "sampling/importance_sampling_ratio/min": 0.22520792484283447, "sampling/sampling_logp_difference/max": 1.4907312393188477, "sampling/sampling_logp_difference/mean": 0.1701533943414688, "step": 318 }, { "clip_ratio/high_max": 0.035777142737060785, "clip_ratio/high_mean": 0.035777142737060785, "clip_ratio/low_mean": 0.014423076994717121, "clip_ratio/low_min": 0.014423076994717121, "clip_ratio/region_mean": 0.050200219731777906, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 213.0, "completions/mean_length": 309.75, "completions/mean_terminated_length": 188.40000915527344, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.8439691588282585, "epoch": 0.012318504788384307, "frac_reward_zero_std": 0.0, "grad_norm": 2.1287405490875244, "learning_rate": 9.036363636363638e-06, "loss": -0.1669, "num_tokens": 684570.0, "reward": 0.6160376071929932, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.14880475401878357, "reward_meter_mean": 0.6731476783752441, "reward_meter_std": 0.4353531301021576, "reward_std": 0.4165080487728119, "reward_total_composite_mean": 0.6160376071929932, "reward_total_composite_std": 0.4165080785751343, "reward_total_mean": 0.6160376071929932, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.14880475401878357, "rewards/meter/mean": 0.6731476783752441, "rewards/meter/std": 0.4353531301021576, "rewards/total_composite/mean": 0.6160376071929932, "rewards/total_composite/std": 0.4165080785751343, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0229874849319458, "sampling/importance_sampling_ratio/min": 0.23035407066345215, "sampling/sampling_logp_difference/max": 1.4681377410888672, "sampling/sampling_logp_difference/mean": 0.11113591492176056, "step": 319 }, { "clip_ratio/high_max": 0.09262609737925231, "clip_ratio/high_mean": 0.09262609737925231, "clip_ratio/low_mean": 0.04342320282012224, "clip_ratio/low_min": 0.04342320282012224, "clip_ratio/region_mean": 0.13604930019937456, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 65.75, "completions/mean_terminated_length": 65.75, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 1.4357076957821846, "epoch": 0.01235712079085573, "frac_reward_zero_std": 0.0, "grad_norm": 8.374490737915039, "learning_rate": 9.033333333333334e-06, "loss": -0.0737, "num_tokens": 686384.0, "reward": 0.7428368330001831, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7547735571861267, "reward_meter_std": 0.39120492339134216, "reward_std": 0.4117808938026428, "reward_total_composite_mean": 0.7428368330001831, "reward_total_composite_std": 0.4117808938026428, "reward_total_mean": 0.7428368330001831, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7547735571861267, "rewards/meter/std": 0.39120492339134216, "rewards/total_composite/mean": 0.7428368330001831, "rewards/total_composite/std": 0.4117808938026428, "sampling/importance_sampling_ratio/max": 1.9665664434432983, "sampling/importance_sampling_ratio/mean": 1.0321924686431885, "sampling/importance_sampling_ratio/min": 0.4012553095817566, "sampling/sampling_logp_difference/max": 0.9131574630737305, "sampling/sampling_logp_difference/mean": 0.11435320973396301, "step": 320 }, { "clip_ratio/high_max": 0.05931214243173599, "clip_ratio/high_mean": 0.05931214243173599, "clip_ratio/low_mean": 0.029992205556482077, "clip_ratio/low_min": 0.029992205556482077, "clip_ratio/region_mean": 0.08930434798821807, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 146.875, "completions/mean_terminated_length": 146.875, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 1.4031646698713303, "epoch": 0.012395736793327155, "frac_reward_zero_std": 0.0, "grad_norm": 5.670385837554932, "learning_rate": 9.030303030303031e-06, "loss": 0.0131, "num_tokens": 688887.0, "reward": 0.8293708562850952, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8293708562850952, "reward_meter_std": 0.18216241896152496, "reward_std": 0.18216243386268616, "reward_total_composite_mean": 0.8293708562850952, "reward_total_composite_std": 0.18216241896152496, "reward_total_mean": 0.8293708562850952, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8293708562850952, "rewards/meter/std": 0.18216241896152496, "rewards/total_composite/mean": 0.8293708562850952, "rewards/total_composite/std": 0.18216241896152496, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.028894305229187, "sampling/importance_sampling_ratio/min": 0.16879475116729736, "sampling/sampling_logp_difference/max": 1.7790718078613281, "sampling/sampling_logp_difference/mean": 0.11904244869947433, "step": 321 }, { "clip_ratio/high_max": 0.10488645080476999, "clip_ratio/high_mean": 0.10488645080476999, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.10488645080476999, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 127.5, "completions/mean_terminated_length": 72.5714340209961, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 1.4858018308877945, "epoch": 0.012434352795798579, "frac_reward_zero_std": 0.0, "grad_norm": 1.3999040126800537, "learning_rate": 9.027272727272728e-06, "loss": -0.1778, "num_tokens": 690651.0, "reward": 0.8692903518676758, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.8722488880157471, "reward_meter_std": 0.34294039011001587, "reward_std": 0.3513069152832031, "reward_total_composite_mean": 0.8692903518676758, "reward_total_composite_std": 0.3513069450855255, "reward_total_mean": 0.8692903518676758, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.8722488880157471, "rewards/meter/std": 0.34294039011001587, "rewards/total_composite/mean": 0.8692903518676758, "rewards/total_composite/std": 0.3513069450855255, "sampling/importance_sampling_ratio/max": 1.8177605867385864, "sampling/importance_sampling_ratio/mean": 1.0229562520980835, "sampling/importance_sampling_ratio/min": 0.33121734857559204, "sampling/sampling_logp_difference/max": 1.10498046875, "sampling/sampling_logp_difference/mean": 0.1413571685552597, "step": 322 }, { "clip_ratio/high_max": 0.05869516870006919, "clip_ratio/high_mean": 0.05869516870006919, "clip_ratio/low_mean": 0.0364344846457243, "clip_ratio/low_min": 0.0364344846457243, "clip_ratio/region_mean": 0.09512965334579349, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 71.25, "completions/mean_terminated_length": 71.25, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 1.1599989607930183, "epoch": 0.012472968798270003, "frac_reward_zero_std": 0.0, "grad_norm": 8.688957214355469, "learning_rate": 9.024242424242426e-06, "loss": -0.0546, "num_tokens": 692669.0, "reward": 0.9870597720146179, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9870597720146179, "reward_meter_std": 0.012377563863992691, "reward_std": 0.012377569451928139, "reward_total_composite_mean": 0.9870597720146179, "reward_total_composite_std": 0.012377563863992691, "reward_total_mean": 0.9870597720146179, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9870597720146179, "rewards/meter/std": 0.012377563863992691, "rewards/total_composite/mean": 0.9870597720146179, "rewards/total_composite/std": 0.012377563863992691, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0185412168502808, "sampling/importance_sampling_ratio/min": 0.3850115239620209, "sampling/sampling_logp_difference/max": 0.9544820785522461, "sampling/sampling_logp_difference/mean": 0.11397970467805862, "step": 323 }, { "clip_ratio/high_max": 0.0074130878783762455, "clip_ratio/high_mean": 0.0074130878783762455, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0074130878783762455, "completions/clipped_ratio": 0.875, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 509.125, "completions/mean_terminated_length": 489.0, "completions/min_length": 489.0, "completions/min_terminated_length": 489.0, "entropy": 0.09911523014307022, "epoch": 0.012511584800741427, "frac_reward_zero_std": 0.0, "grad_norm": 0.7056026458740234, "learning_rate": 9.021212121212121e-06, "loss": -0.1171, "num_tokens": 694598.0, "reward": 0.33366143703460693, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_count_adherence_mean": 0.6180555820465088, "reward_count_adherence_std": 0.20452874898910522, "reward_meter_mean": 0.5128244161605835, "reward_meter_std": 0.40451905131340027, "reward_std": 0.3783239722251892, "reward_total_composite_mean": 0.33366143703460693, "reward_total_composite_std": 0.3783239722251892, "reward_total_mean": 0.33366143703460693, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/count_adherence/mean": 0.6180555820465088, "rewards/count_adherence/std": 0.20452874898910522, "rewards/meter/mean": 0.5128244161605835, "rewards/meter/std": 0.40451905131340027, "rewards/total_composite/mean": 0.33366143703460693, "rewards/total_composite/std": 0.3783239722251892, "sampling/importance_sampling_ratio/max": 1.636110782623291, "sampling/importance_sampling_ratio/mean": 1.012312650680542, "sampling/importance_sampling_ratio/min": 0.40183717012405396, "sampling/sampling_logp_difference/max": 0.9117083549499512, "sampling/sampling_logp_difference/mean": 0.07055795192718506, "step": 324 }, { "clip_ratio/high_max": 0.07463606260716915, "clip_ratio/high_mean": 0.07463606260716915, "clip_ratio/low_mean": 0.027904992923140526, "clip_ratio/low_min": 0.027904992923140526, "clip_ratio/region_mean": 0.10254105553030968, "completions/clipped_ratio": 0.0, "completions/max_length": 192.0, "completions/max_terminated_length": 192.0, "completions/mean_length": 173.875, "completions/mean_terminated_length": 173.875, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "entropy": 1.7032774537801743, "epoch": 0.012550200803212851, "frac_reward_zero_std": 0.0, "grad_norm": 5.355256080627441, "learning_rate": 9.01818181818182e-06, "loss": -0.0167, "num_tokens": 697565.0, "reward": 0.7156921625137329, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_meter_mean": 0.7737569212913513, "reward_meter_std": 0.38154885172843933, "reward_std": 0.37208291888237, "reward_total_composite_mean": 0.7156921625137329, "reward_total_composite_std": 0.37208291888237, "reward_total_mean": 0.7156921625137329, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/meter/mean": 0.7737569212913513, "rewards/meter/std": 0.38154885172843933, "rewards/total_composite/mean": 0.7156921625137329, "rewards/total_composite/std": 0.37208291888237, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0292820930480957, "sampling/importance_sampling_ratio/min": 0.14826318621635437, "sampling/sampling_logp_difference/max": 1.908766269683838, "sampling/sampling_logp_difference/mean": 0.13555942475795746, "step": 325 }, { "clip_ratio/high_max": 0.019201413029804826, "clip_ratio/high_mean": 0.019201413029804826, "clip_ratio/low_mean": 0.00886194035410881, "clip_ratio/low_min": 0.00886194035410881, "clip_ratio/region_mean": 0.028063353383913636, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 340.375, "completions/mean_terminated_length": 283.16668701171875, "completions/min_length": 250.0, "completions/min_terminated_length": 250.0, "entropy": 0.34993776120245457, "epoch": 0.012588816805684275, "frac_reward_zero_std": 0.0, "grad_norm": 0.9641727209091187, "learning_rate": 9.015151515151516e-06, "loss": -0.3057, "num_tokens": 700832.0, "reward": 0.642575740814209, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.2531938850879669, "reward_meter_mean": 0.9441099166870117, "reward_meter_std": 0.12238472700119019, "reward_std": 0.2958056628704071, "reward_total_composite_mean": 0.642575740814209, "reward_total_composite_std": 0.2958056926727295, "reward_total_mean": 0.642575740814209, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.2531938850879669, "rewards/meter/mean": 0.9441099166870117, "rewards/meter/std": 0.12238472700119019, "rewards/total_composite/mean": 0.642575740814209, "rewards/total_composite/std": 0.2958056926727295, "sampling/importance_sampling_ratio/max": 1.7436391115188599, "sampling/importance_sampling_ratio/mean": 1.008855938911438, "sampling/importance_sampling_ratio/min": 0.35944268107414246, "sampling/sampling_logp_difference/max": 1.0232006311416626, "sampling/sampling_logp_difference/mean": 0.04643620178103447, "step": 326 }, { "clip_ratio/high_max": 0.03814008738845587, "clip_ratio/high_mean": 0.03814008738845587, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.03814008738845587, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 309.625, "completions/mean_terminated_length": 280.71429443359375, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "entropy": 0.48327015712857246, "epoch": 0.0126274328081557, "frac_reward_zero_std": 0.0, "grad_norm": 0.8649334907531738, "learning_rate": 9.012121212121213e-06, "loss": -0.2768, "num_tokens": 704333.0, "reward": 0.7233796119689941, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7361111640930176, "reward_count_adherence_std": 0.25845491886138916, "reward_meter_mean": 0.9838745594024658, "reward_meter_std": 0.026461318135261536, "reward_std": 0.252864271402359, "reward_total_composite_mean": 0.7233796119689941, "reward_total_composite_std": 0.2528642416000366, "reward_total_mean": 0.7233796119689941, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7361111640930176, "rewards/count_adherence/std": 0.25845491886138916, "rewards/meter/mean": 0.9838745594024658, "rewards/meter/std": 0.026461318135261536, "rewards/total_composite/mean": 0.7233796119689941, "rewards/total_composite/std": 0.2528642416000366, "sampling/importance_sampling_ratio/max": 1.7859697341918945, "sampling/importance_sampling_ratio/mean": 1.010648488998413, "sampling/importance_sampling_ratio/min": 0.17294643819332123, "sampling/sampling_logp_difference/max": 1.7547733783721924, "sampling/sampling_logp_difference/mean": 0.05245785042643547, "step": 327 }, { "clip_ratio/high_max": 0.11218475922942162, "clip_ratio/high_mean": 0.11218475922942162, "clip_ratio/low_mean": 0.05930484738200903, "clip_ratio/low_min": 0.05930484738200903, "clip_ratio/region_mean": 0.17148960661143064, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 34.125, "completions/mean_terminated_length": 34.125, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 1.6025570929050446, "epoch": 0.012666048810627124, "frac_reward_zero_std": 0.0, "grad_norm": 11.575224876403809, "learning_rate": 9.00909090909091e-06, "loss": 0.0311, "num_tokens": 705878.0, "reward": 0.7254409790039062, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7254409790039062, "reward_meter_std": 0.31982889771461487, "reward_std": 0.31982889771461487, "reward_total_composite_mean": 0.7254409790039062, "reward_total_composite_std": 0.31982889771461487, "reward_total_mean": 0.7254409790039062, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7254409790039062, "rewards/meter/std": 0.31982889771461487, "rewards/total_composite/mean": 0.7254409790039062, "rewards/total_composite/std": 0.31982889771461487, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0245475769042969, "sampling/importance_sampling_ratio/min": 0.2002735137939453, "sampling/sampling_logp_difference/max": 1.6080713272094727, "sampling/sampling_logp_difference/mean": 0.16250382363796234, "step": 328 }, { "clip_ratio/high_max": 0.07188303023576736, "clip_ratio/high_mean": 0.07188303023576736, "clip_ratio/low_mean": 0.024105392396450043, "clip_ratio/low_min": 0.024105392396450043, "clip_ratio/region_mean": 0.09598842263221741, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 154.0, "completions/mean_length": 193.125, "completions/mean_terminated_length": 147.57144165039062, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "entropy": 1.2332377284765244, "epoch": 0.012704664813098548, "frac_reward_zero_std": 0.0, "grad_norm": 4.62138032913208, "learning_rate": 9.006060606060607e-06, "loss": 0.0049, "num_tokens": 708351.0, "reward": 0.7189717292785645, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.7495023012161255, "reward_meter_std": 0.3558323085308075, "reward_std": 0.3438011407852173, "reward_total_composite_mean": 0.7189717292785645, "reward_total_composite_std": 0.3438011407852173, "reward_total_mean": 0.7189717292785645, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.7495023012161255, "rewards/meter/std": 0.3558323085308075, "rewards/total_composite/mean": 0.7189717292785645, "rewards/total_composite/std": 0.3438011407852173, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0160291194915771, "sampling/importance_sampling_ratio/min": 0.3046167492866516, "sampling/sampling_logp_difference/max": 1.1887009143829346, "sampling/sampling_logp_difference/mean": 0.12337387353181839, "step": 329 }, { "clip_ratio/high_max": 0.10720423050224781, "clip_ratio/high_mean": 0.10720423050224781, "clip_ratio/low_mean": 0.025510204955935478, "clip_ratio/low_min": 0.025510204955935478, "clip_ratio/region_mean": 0.1327144354581833, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 113.25, "completions/mean_terminated_length": 56.28571701049805, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 1.7517335712909698, "epoch": 0.012743280815569972, "frac_reward_zero_std": 0.0, "grad_norm": 2.0594024658203125, "learning_rate": 9.003030303030303e-06, "loss": -0.155, "num_tokens": 709897.0, "reward": 0.8302444815635681, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8331777453422546, "reward_meter_std": 0.3333028256893158, "reward_std": 0.34145045280456543, "reward_total_composite_mean": 0.8302444815635681, "reward_total_composite_std": 0.34145042300224304, "reward_total_mean": 0.8302444815635681, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8331777453422546, "rewards/meter/std": 0.3333028256893158, "rewards/total_composite/mean": 0.8302444815635681, "rewards/total_composite/std": 0.34145042300224304, "sampling/importance_sampling_ratio/max": 1.7736172676086426, "sampling/importance_sampling_ratio/mean": 1.0383025407791138, "sampling/importance_sampling_ratio/min": 0.07492611557245255, "sampling/sampling_logp_difference/max": 2.5912528038024902, "sampling/sampling_logp_difference/mean": 0.1602991372346878, "step": 330 }, { "clip_ratio/high_max": 0.06826505670323968, "clip_ratio/high_mean": 0.06826505670323968, "clip_ratio/low_mean": 0.03799927420914173, "clip_ratio/low_min": 0.03799927420914173, "clip_ratio/region_mean": 0.10626433091238141, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 72.375, "completions/mean_terminated_length": 72.375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.6253799833357334, "epoch": 0.012781896818041396, "frac_reward_zero_std": 0.0, "grad_norm": 19.114912033081055, "learning_rate": 9e-06, "loss": 0.1271, "num_tokens": 712092.0, "reward": 0.665668785572052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.7024174928665161, "reward_meter_std": 0.4215409457683563, "reward_std": 0.41643577814102173, "reward_total_composite_mean": 0.665668785572052, "reward_total_composite_std": 0.41643577814102173, "reward_total_mean": 0.665668785572052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.7024174928665161, "rewards/meter/std": 0.4215409457683563, "rewards/total_composite/mean": 0.665668785572052, "rewards/total_composite/std": 0.41643577814102173, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9930925965309143, "sampling/importance_sampling_ratio/min": 0.15829293429851532, "sampling/sampling_logp_difference/max": 1.8433079719543457, "sampling/sampling_logp_difference/mean": 0.12084802985191345, "step": 331 }, { "clip_ratio/high_max": 0.05861414410173893, "clip_ratio/high_mean": 0.05861414410173893, "clip_ratio/low_mean": 0.11972531583160162, "clip_ratio/low_min": 0.11972531583160162, "clip_ratio/region_mean": 0.17833945993334055, "completions/clipped_ratio": 0.0, "completions/max_length": 49.0, "completions/max_terminated_length": 49.0, "completions/mean_length": 40.375, "completions/mean_terminated_length": 40.375, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 1.166979692876339, "epoch": 0.01282051282051282, "frac_reward_zero_std": 0.0, "grad_norm": 17.004131317138672, "learning_rate": 8.996969696969697e-06, "loss": -0.0274, "num_tokens": 713551.0, "reward": 0.4543929100036621, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4543929100036621, "reward_meter_std": 0.32974186539649963, "reward_std": 0.32974183559417725, "reward_total_composite_mean": 0.4543929100036621, "reward_total_composite_std": 0.32974186539649963, "reward_total_mean": 0.4543929100036621, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4543929100036621, "rewards/meter/std": 0.32974186539649963, "rewards/total_composite/mean": 0.4543929100036621, "rewards/total_composite/std": 0.32974186539649963, "sampling/importance_sampling_ratio/max": 1.9088724851608276, "sampling/importance_sampling_ratio/mean": 1.0065240859985352, "sampling/importance_sampling_ratio/min": 0.05844109505414963, "sampling/sampling_logp_difference/max": 2.839735984802246, "sampling/sampling_logp_difference/mean": 0.1603153944015503, "step": 332 }, { "clip_ratio/high_max": 0.04428324103355408, "clip_ratio/high_mean": 0.04428324103355408, "clip_ratio/low_mean": 0.024796965066343546, "clip_ratio/low_min": 0.024796965066343546, "clip_ratio/region_mean": 0.06908020609989762, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 87.625, "completions/mean_terminated_length": 87.625, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.8614702522754669, "epoch": 0.012859128822984244, "frac_reward_zero_std": 0.0, "grad_norm": 5.926055908203125, "learning_rate": 8.993939393939395e-06, "loss": 0.0326, "num_tokens": 715540.0, "reward": 0.8005719184875488, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8005719184875488, "reward_meter_std": 0.2811793088912964, "reward_std": 0.2811793386936188, "reward_total_composite_mean": 0.8005719184875488, "reward_total_composite_std": 0.2811793088912964, "reward_total_mean": 0.8005719184875488, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8005719184875488, "rewards/meter/std": 0.2811793088912964, "rewards/total_composite/mean": 0.8005719184875488, "rewards/total_composite/std": 0.2811793088912964, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.023775577545166, "sampling/importance_sampling_ratio/min": 0.12364701926708221, "sampling/sampling_logp_difference/max": 2.0903244018554688, "sampling/sampling_logp_difference/mean": 0.093291737139225, "step": 333 }, { "clip_ratio/high_max": 0.13734323251992464, "clip_ratio/high_mean": 0.13734323251992464, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/region_mean": 0.15296823251992464, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 38.25, "completions/mean_terminated_length": 38.25, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 1.5458678156137466, "epoch": 0.012897744825455668, "frac_reward_zero_std": 0.0, "grad_norm": 14.146446228027344, "learning_rate": 8.990909090909092e-06, "loss": 0.0292, "num_tokens": 717022.0, "reward": 0.9717813730239868, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9717813730239868, "reward_meter_std": 0.05659351125359535, "reward_std": 0.05659349635243416, "reward_total_composite_mean": 0.9717813730239868, "reward_total_composite_std": 0.05659351125359535, "reward_total_mean": 0.9717813730239868, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9717813730239868, "rewards/meter/std": 0.05659351125359535, "rewards/total_composite/mean": 0.9717813730239868, "rewards/total_composite/std": 0.05659351125359535, "sampling/importance_sampling_ratio/max": 1.9646681547164917, "sampling/importance_sampling_ratio/mean": 1.0397027730941772, "sampling/importance_sampling_ratio/min": 0.33652088046073914, "sampling/sampling_logp_difference/max": 1.089095115661621, "sampling/sampling_logp_difference/mean": 0.1425103396177292, "step": 334 }, { "clip_ratio/high_max": 0.10228652320802212, "clip_ratio/high_mean": 0.10228652320802212, "clip_ratio/low_mean": 0.01715686358511448, "clip_ratio/low_min": 0.01715686358511448, "clip_ratio/region_mean": 0.1194433867931366, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 63.25, "completions/mean_terminated_length": 63.25, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 1.2129319533705711, "epoch": 0.012936360827927092, "frac_reward_zero_std": 0.0, "grad_norm": 7.787652969360352, "learning_rate": 8.98787878787879e-06, "loss": -0.0594, "num_tokens": 718816.0, "reward": 0.8696156740188599, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8696156740188599, "reward_meter_std": 0.31520524621009827, "reward_std": 0.31520524621009827, "reward_total_composite_mean": 0.8696156740188599, "reward_total_composite_std": 0.31520524621009827, "reward_total_mean": 0.8696156740188599, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8696156740188599, "rewards/meter/std": 0.31520524621009827, "rewards/total_composite/mean": 0.8696156740188599, "rewards/total_composite/std": 0.31520524621009827, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0193424224853516, "sampling/importance_sampling_ratio/min": 0.32608509063720703, "sampling/sampling_logp_difference/max": 1.1205968856811523, "sampling/sampling_logp_difference/mean": 0.13351094722747803, "step": 335 }, { "clip_ratio/high_max": 0.04853037279099226, "clip_ratio/high_mean": 0.04853037279099226, "clip_ratio/low_mean": 0.027916074730455875, "clip_ratio/low_min": 0.027916074730455875, "clip_ratio/region_mean": 0.07644644752144814, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 67.75, "completions/mean_terminated_length": 67.75, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.7481025010347366, "epoch": 0.012974976830398516, "frac_reward_zero_std": 0.0, "grad_norm": 7.147242069244385, "learning_rate": 8.984848484848485e-06, "loss": 0.0228, "num_tokens": 720702.0, "reward": 0.9919648766517639, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9919648766517639, "reward_meter_std": 0.00817751046270132, "reward_std": 0.008177503012120724, "reward_total_composite_mean": 0.9919648766517639, "reward_total_composite_std": 0.00817751046270132, "reward_total_mean": 0.9919648766517639, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9919648766517639, "rewards/meter/std": 0.00817751046270132, "rewards/total_composite/mean": 0.9919648766517639, "rewards/total_composite/std": 0.00817751046270132, "sampling/importance_sampling_ratio/max": 1.912467360496521, "sampling/importance_sampling_ratio/mean": 1.0083116292953491, "sampling/importance_sampling_ratio/min": 0.08475096523761749, "sampling/sampling_logp_difference/max": 2.4680380821228027, "sampling/sampling_logp_difference/mean": 0.10161230713129044, "step": 336 }, { "clip_ratio/high_max": 0.07585466234013438, "clip_ratio/high_mean": 0.07585466234013438, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/region_mean": 0.08320760354399681, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 85.0, "completions/mean_terminated_length": 85.0, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.6667735129594803, "epoch": 0.01301359283286994, "frac_reward_zero_std": 0.0, "grad_norm": 6.292624473571777, "learning_rate": 8.981818181818182e-06, "loss": 0.3588, "num_tokens": 722606.0, "reward": 0.8667486310005188, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.9909840822219849, "reward_meter_std": 0.0034961404744535685, "reward_std": 0.35023483633995056, "reward_total_composite_mean": 0.8667486310005188, "reward_total_composite_std": 0.35023483633995056, "reward_total_mean": 0.8667486310005188, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.9909840822219849, "rewards/meter/std": 0.0034961404744535685, "rewards/total_composite/mean": 0.8667486310005188, "rewards/total_composite/std": 0.35023483633995056, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.014748454093933, "sampling/importance_sampling_ratio/min": 0.18162985146045685, "sampling/sampling_logp_difference/max": 1.7057844400405884, "sampling/sampling_logp_difference/mean": 0.07967483252286911, "step": 337 }, { "clip_ratio/high_max": 0.02921690931543708, "clip_ratio/high_mean": 0.02921690931543708, "clip_ratio/low_mean": 0.027584444032981992, "clip_ratio/low_min": 0.027584444032981992, "clip_ratio/region_mean": 0.05680135334841907, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 94.125, "completions/mean_terminated_length": 94.125, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.38130839727818966, "epoch": 0.013052208835341365, "frac_reward_zero_std": 0.0, "grad_norm": 6.0476861000061035, "learning_rate": 8.97878787878788e-06, "loss": 0.0094, "num_tokens": 724687.0, "reward": 0.7916698455810547, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7916698455810547, "reward_meter_std": 0.20429320633411407, "reward_std": 0.2042931765317917, "reward_total_composite_mean": 0.7916698455810547, "reward_total_composite_std": 0.20429320633411407, "reward_total_mean": 0.7916698455810547, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7916698455810547, "rewards/meter/std": 0.20429320633411407, "rewards/total_composite/mean": 0.7916698455810547, "rewards/total_composite/std": 0.20429320633411407, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0018184185028076, "sampling/importance_sampling_ratio/min": 0.2368934005498886, "sampling/sampling_logp_difference/max": 1.4401450157165527, "sampling/sampling_logp_difference/mean": 0.051787663251161575, "step": 338 }, { "clip_ratio/high_max": 0.06074862740933895, "clip_ratio/high_mean": 0.06074862740933895, "clip_ratio/low_mean": 0.0517513370141387, "clip_ratio/low_min": 0.0517513370141387, "clip_ratio/region_mean": 0.11249996442347765, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 25.5, "completions/mean_terminated_length": 25.5, "completions/min_length": 20.0, "completions/min_terminated_length": 20.0, "entropy": 0.8039275705814362, "epoch": 0.013090824837812789, "frac_reward_zero_std": 0.0, "grad_norm": 32.10745620727539, "learning_rate": 8.975757575757577e-06, "loss": 0.1049, "num_tokens": 726091.0, "reward": 0.45373764634132385, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.45373764634132385, "reward_meter_std": 0.3389832377433777, "reward_std": 0.3389832377433777, "reward_total_composite_mean": 0.45373764634132385, "reward_total_composite_std": 0.3389832377433777, "reward_total_mean": 0.45373764634132385, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.45373764634132385, "rewards/meter/std": 0.3389832377433777, "rewards/total_composite/mean": 0.45373764634132385, "rewards/total_composite/std": 0.3389832377433777, "sampling/importance_sampling_ratio/max": 1.9982775449752808, "sampling/importance_sampling_ratio/mean": 0.9738427400588989, "sampling/importance_sampling_ratio/min": 0.186411514878273, "sampling/sampling_logp_difference/max": 1.6797986030578613, "sampling/sampling_logp_difference/mean": 0.163490429520607, "step": 339 }, { "clip_ratio/high_max": 0.05819632112979889, "clip_ratio/high_mean": 0.05819632112979889, "clip_ratio/low_mean": 0.03718182072043419, "clip_ratio/low_min": 0.03718182072043419, "clip_ratio/region_mean": 0.09537814185023308, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 181.0, "completions/mean_length": 206.875, "completions/mean_terminated_length": 163.2857208251953, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.5428336411714554, "epoch": 0.013129440840284215, "frac_reward_zero_std": 0.0, "grad_norm": 4.418445110321045, "learning_rate": 8.972727272727272e-06, "loss": -0.0956, "num_tokens": 728882.0, "reward": 0.39793068170547485, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.3452087938785553, "reward_meter_mean": 0.4240845739841461, "reward_meter_std": 0.30786141753196716, "reward_std": 0.29285892844200134, "reward_total_composite_mean": 0.39793068170547485, "reward_total_composite_std": 0.29285889863967896, "reward_total_mean": 0.39793068170547485, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.3452087938785553, "rewards/meter/mean": 0.4240845739841461, "rewards/meter/std": 0.30786141753196716, "rewards/total_composite/mean": 0.39793068170547485, "rewards/total_composite/std": 0.29285889863967896, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0121982097625732, "sampling/importance_sampling_ratio/min": 0.04704439640045166, "sampling/sampling_logp_difference/max": 3.0566635131835938, "sampling/sampling_logp_difference/mean": 0.10585647076368332, "step": 340 }, { "clip_ratio/high_max": 0.11759159062057734, "clip_ratio/high_mean": 0.11759159062057734, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.11759159062057734, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 54.5, "completions/mean_terminated_length": 54.5, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 1.2029671669006348, "epoch": 0.013168056842755639, "frac_reward_zero_std": 0.0, "grad_norm": 7.589391708374023, "learning_rate": 8.969696969696971e-06, "loss": -0.1819, "num_tokens": 730886.0, "reward": 0.9168034195899963, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9767169952392578, "reward_meter_std": 0.013271527364850044, "reward_std": 0.17712122201919556, "reward_total_composite_mean": 0.9168034195899963, "reward_total_composite_std": 0.17712123692035675, "reward_total_mean": 0.9168034195899963, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9767169952392578, "rewards/meter/std": 0.013271527364850044, "rewards/total_composite/mean": 0.9168034195899963, "rewards/total_composite/std": 0.17712123692035675, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.022121787071228, "sampling/importance_sampling_ratio/min": 0.33108022809028625, "sampling/sampling_logp_difference/max": 1.1053946018218994, "sampling/sampling_logp_difference/mean": 0.12854255735874176, "step": 341 }, { "clip_ratio/high_max": 0.08128050155937672, "clip_ratio/high_mean": 0.08128050155937672, "clip_ratio/low_mean": 0.021226415410637856, "clip_ratio/low_min": 0.021226415410637856, "clip_ratio/region_mean": 0.10250691697001457, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 146.5, "completions/mean_terminated_length": 94.28572082519531, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 1.2604146748781204, "epoch": 0.013206672845227063, "frac_reward_zero_std": 0.0, "grad_norm": 4.1543121337890625, "learning_rate": 8.966666666666667e-06, "loss": -0.1304, "num_tokens": 733050.0, "reward": 0.7020714282989502, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7078612446784973, "reward_meter_std": 0.3426123261451721, "reward_std": 0.3555365204811096, "reward_total_composite_mean": 0.7020714282989502, "reward_total_composite_std": 0.3555365204811096, "reward_total_mean": 0.7020714282989502, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7078612446784973, "rewards/meter/std": 0.3426123261451721, "rewards/total_composite/mean": 0.7020714282989502, "rewards/total_composite/std": 0.3555365204811096, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.020814299583435, "sampling/importance_sampling_ratio/min": 0.20932936668395996, "sampling/sampling_logp_difference/max": 1.5638463497161865, "sampling/sampling_logp_difference/mean": 0.15139326453208923, "step": 342 }, { "clip_ratio/high_max": 0.02198702801251784, "clip_ratio/high_mean": 0.02198702801251784, "clip_ratio/low_mean": 0.007872846443206072, "clip_ratio/low_min": 0.007872846443206072, "clip_ratio/region_mean": 0.02985987445572391, "completions/clipped_ratio": 0.0, "completions/max_length": 162.0, "completions/max_terminated_length": 162.0, "completions/mean_length": 136.75, "completions/mean_terminated_length": 136.75, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.2882226835936308, "epoch": 0.013245288847698487, "frac_reward_zero_std": 0.0, "grad_norm": 6.76770544052124, "learning_rate": 8.963636363636364e-06, "loss": -0.0077, "num_tokens": 735640.0, "reward": 0.7804628610610962, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_meter_mean": 0.866592288017273, "reward_meter_std": 0.23036333918571472, "reward_std": 0.22920989990234375, "reward_total_composite_mean": 0.7804628610610962, "reward_total_composite_std": 0.22920989990234375, "reward_total_mean": 0.7804628610610962, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/meter/mean": 0.866592288017273, "rewards/meter/std": 0.23036333918571472, "rewards/total_composite/mean": 0.7804628610610962, "rewards/total_composite/std": 0.22920989990234375, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0099705457687378, "sampling/importance_sampling_ratio/min": 0.0917281061410904, "sampling/sampling_logp_difference/max": 2.3889265060424805, "sampling/sampling_logp_difference/mean": 0.04342114180326462, "step": 343 }, { "clip_ratio/high_max": 0.001968626049347222, "clip_ratio/high_mean": 0.001968626049347222, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.001968626049347222, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 511.0, "completions/mean_terminated_length": 508.0, "completions/min_length": 504.0, "completions/min_terminated_length": 504.0, "entropy": 0.015514392405748367, "epoch": 0.013283904850169911, "frac_reward_zero_std": 0.0, "grad_norm": 0.3268384635448456, "learning_rate": 8.960606060606061e-06, "loss": -0.1251, "num_tokens": 738392.0, "reward": 0.6975077390670776, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.736842155456543, "reward_count_adherence_std": 0.06891091912984848, "reward_meter_mean": 0.9466683864593506, "reward_meter_std": 0.14325150847434998, "reward_std": 0.12571200728416443, "reward_total_composite_mean": 0.6975077390670776, "reward_total_composite_std": 0.12571200728416443, "reward_total_mean": 0.6975077390670776, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.736842155456543, "rewards/count_adherence/std": 0.06891091912984848, "rewards/meter/mean": 0.9466683864593506, "rewards/meter/std": 0.14325150847434998, "rewards/total_composite/mean": 0.6975077390670776, "rewards/total_composite/std": 0.12571200728416443, "sampling/importance_sampling_ratio/max": 1.704252004623413, "sampling/importance_sampling_ratio/mean": 1.0004547834396362, "sampling/importance_sampling_ratio/min": 0.46806204319000244, "sampling/sampling_logp_difference/max": 0.7591544389724731, "sampling/sampling_logp_difference/mean": 0.007342774420976639, "step": 344 }, { "clip_ratio/high_max": 0.049760105554014444, "clip_ratio/high_mean": 0.049760105554014444, "clip_ratio/low_mean": 0.02456349227577448, "clip_ratio/low_min": 0.02456349227577448, "clip_ratio/region_mean": 0.07432359782978892, "completions/clipped_ratio": 0.0, "completions/max_length": 151.0, "completions/max_terminated_length": 151.0, "completions/mean_length": 135.625, "completions/mean_terminated_length": 135.625, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.7220766767859459, "epoch": 0.013322520852641335, "frac_reward_zero_std": 0.0, "grad_norm": 7.332951068878174, "learning_rate": 8.957575757575758e-06, "loss": 0.0533, "num_tokens": 740877.0, "reward": 0.8432111740112305, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8432111740112305, "reward_meter_std": 0.3098846971988678, "reward_std": 0.3098846971988678, "reward_total_composite_mean": 0.8432111740112305, "reward_total_composite_std": 0.3098846971988678, "reward_total_mean": 0.8432111740112305, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8432111740112305, "rewards/meter/std": 0.3098846971988678, "rewards/total_composite/mean": 0.8432111740112305, "rewards/total_composite/std": 0.3098846971988678, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0232460498809814, "sampling/importance_sampling_ratio/min": 0.20046362280845642, "sampling/sampling_logp_difference/max": 1.6071224212646484, "sampling/sampling_logp_difference/mean": 0.09029541909694672, "step": 345 }, { "clip_ratio/high_max": 0.060007378458976746, "clip_ratio/high_mean": 0.060007378458976746, "clip_ratio/low_mean": 0.025923453271389008, "clip_ratio/low_min": 0.025923453271389008, "clip_ratio/region_mean": 0.08593083173036575, "completions/clipped_ratio": 0.0, "completions/max_length": 175.0, "completions/max_terminated_length": 175.0, "completions/mean_length": 136.25, "completions/mean_terminated_length": 136.25, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.7361192889511585, "epoch": 0.01336113685511276, "frac_reward_zero_std": 0.0, "grad_norm": 6.021731853485107, "learning_rate": 8.954545454545456e-06, "loss": -0.0646, "num_tokens": 743391.0, "reward": 0.8351510763168335, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.8487316966056824, "reward_meter_std": 0.22170411050319672, "reward_std": 0.2519603669643402, "reward_total_composite_mean": 0.8351510763168335, "reward_total_composite_std": 0.2519603967666626, "reward_total_mean": 0.8351510763168335, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.8487316966056824, "rewards/meter/std": 0.22170411050319672, "rewards/total_composite/mean": 0.8351510763168335, "rewards/total_composite/std": 0.2519603967666626, "sampling/importance_sampling_ratio/max": 1.87782621383667, "sampling/importance_sampling_ratio/mean": 1.0095629692077637, "sampling/importance_sampling_ratio/min": 0.2516863942146301, "sampling/sampling_logp_difference/max": 1.3795714378356934, "sampling/sampling_logp_difference/mean": 0.08980220556259155, "step": 346 }, { "clip_ratio/high_max": 0.0906870374456048, "clip_ratio/high_mean": 0.0906870374456048, "clip_ratio/low_mean": 0.008223684504628181, "clip_ratio/low_min": 0.008223684504628181, "clip_ratio/region_mean": 0.09891072195023298, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 84.0, "completions/mean_terminated_length": 84.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 1.0877859592437744, "epoch": 0.013399752857584183, "frac_reward_zero_std": 0.0, "grad_norm": 10.403669357299805, "learning_rate": 8.951515151515153e-06, "loss": 0.3161, "num_tokens": 745255.0, "reward": 0.8664380311965942, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.9903609752655029, "reward_meter_std": 0.0039792293682694435, "reward_std": 0.35011619329452515, "reward_total_composite_mean": 0.8664380311965942, "reward_total_composite_std": 0.35011622309684753, "reward_total_mean": 0.8664380311965942, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.9903609752655029, "rewards/meter/std": 0.0039792293682694435, "rewards/total_composite/mean": 0.8664380311965942, "rewards/total_composite/std": 0.35011622309684753, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0305798053741455, "sampling/importance_sampling_ratio/min": 0.31438201665878296, "sampling/sampling_logp_difference/max": 1.1571464538574219, "sampling/sampling_logp_difference/mean": 0.09673066437244415, "step": 347 }, { "clip_ratio/high_max": 0.035924003925174475, "clip_ratio/high_mean": 0.035924003925174475, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/region_mean": 0.04728764062747359, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 261.0, "completions/mean_length": 268.5, "completions/mean_terminated_length": 233.71429443359375, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.3383890204131603, "epoch": 0.013438368860055607, "frac_reward_zero_std": 0.0, "grad_norm": 1.5081286430358887, "learning_rate": 8.94848484848485e-06, "loss": -0.2579, "num_tokens": 748739.0, "reward": 0.7942942976951599, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.24147263169288635, "reward_meter_mean": 0.8488904237747192, "reward_meter_std": 0.3285263478755951, "reward_std": 0.33547642827033997, "reward_total_composite_mean": 0.7942942976951599, "reward_total_composite_std": 0.33547642827033997, "reward_total_mean": 0.7942942976951599, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.24147263169288635, "rewards/meter/mean": 0.8488904237747192, "rewards/meter/std": 0.3285263478755951, "rewards/total_composite/mean": 0.7942942976951599, "rewards/total_composite/std": 0.33547642827033997, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0065252780914307, "sampling/importance_sampling_ratio/min": 0.0667203888297081, "sampling/sampling_logp_difference/max": 2.707244634628296, "sampling/sampling_logp_difference/mean": 0.05050771310925484, "step": 348 }, { "clip_ratio/high_max": 0.07814110023900867, "clip_ratio/high_mean": 0.07814110023900867, "clip_ratio/low_mean": 0.025058357510715723, "clip_ratio/low_min": 0.025058357510715723, "clip_ratio/region_mean": 0.10319945774972439, "completions/clipped_ratio": 0.0, "completions/max_length": 153.0, "completions/max_terminated_length": 153.0, "completions/mean_length": 125.25, "completions/mean_terminated_length": 125.25, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.955632783472538, "epoch": 0.013476984862527032, "frac_reward_zero_std": 0.0, "grad_norm": 6.383397102355957, "learning_rate": 8.945454545454546e-06, "loss": 0.0873, "num_tokens": 751125.0, "reward": 0.974650502204895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.974650502204895, "reward_meter_std": 0.015549466013908386, "reward_std": 0.01554945856332779, "reward_total_composite_mean": 0.974650502204895, "reward_total_composite_std": 0.015549466013908386, "reward_total_mean": 0.974650502204895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.974650502204895, "rewards/meter/std": 0.015549466013908386, "rewards/total_composite/mean": 0.974650502204895, "rewards/total_composite/std": 0.015549466013908386, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01600182056427, "sampling/importance_sampling_ratio/min": 0.2371901124715805, "sampling/sampling_logp_difference/max": 1.4388933181762695, "sampling/sampling_logp_difference/mean": 0.1007874459028244, "step": 349 }, { "clip_ratio/high_max": 0.07839446794241667, "clip_ratio/high_mean": 0.07839446794241667, "clip_ratio/low_mean": 0.044384559616446495, "clip_ratio/low_min": 0.044384559616446495, "clip_ratio/region_mean": 0.12277902755886316, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 35.875, "completions/mean_terminated_length": 35.875, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 1.2592712044715881, "epoch": 0.013515600864998456, "frac_reward_zero_std": 0.0, "grad_norm": 14.133708953857422, "learning_rate": 8.942424242424243e-06, "loss": -0.018, "num_tokens": 752644.0, "reward": 0.690624475479126, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.690624475479126, "reward_meter_std": 0.37756747007369995, "reward_std": 0.37756747007369995, "reward_total_composite_mean": 0.690624475479126, "reward_total_composite_std": 0.37756747007369995, "reward_total_mean": 0.690624475479126, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.690624475479126, "rewards/meter/std": 0.37756747007369995, "rewards/total_composite/mean": 0.690624475479126, "rewards/total_composite/std": 0.37756747007369995, "sampling/importance_sampling_ratio/max": 1.7945036888122559, "sampling/importance_sampling_ratio/mean": 1.0118988752365112, "sampling/importance_sampling_ratio/min": 0.24649037420749664, "sampling/sampling_logp_difference/max": 1.4004323482513428, "sampling/sampling_logp_difference/mean": 0.1355486959218979, "step": 350 }, { "epoch": 0.013515600864998456, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.038461538461538464, "eval_completions/max_length": 442.84615384615387, "eval_completions/max_terminated_length": 400.0, "eval_completions/mean_length": 212.46153846153845, "eval_completions/mean_terminated_length": 200.26236431415265, "eval_completions/min_length": 56.0, "eval_completions/min_terminated_length": 56.0, "eval_entropy": 0.43929139238137466, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 752644.0, "eval_reward": 0.6290297783338107, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.8791503814550546, "eval_reward_count_adherence_std": 0.17575800361541602, "eval_reward_meter_mean": 0.7184457114109626, "eval_reward_meter_std": 0.3159666336499728, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6290297783338107, "eval_reward_total_composite_std": 0.31780216900201946, "eval_reward_total_mean": 0.6290297783338107, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.8791503814550546, "eval_rewards/count_adherence/std": 0.17575800361541602, "eval_rewards/meter/mean": 0.7184457114109626, "eval_rewards/meter/std": 0.3159666336499728, "eval_rewards/total_composite/mean": 0.6290297783338107, "eval_rewards/total_composite/std": 0.31780216900201946, "eval_runtime": 82.5629, "eval_samples_per_second": 1.26, "eval_sampling/importance_sampling_ratio/max": 1.47766217818627, "eval_sampling/importance_sampling_ratio/mean": 1.0086628473722017, "eval_sampling/importance_sampling_ratio/min": 0.3609803124116017, "eval_sampling/sampling_logp_difference/max": 1.040850510964027, "eval_sampling/sampling_logp_difference/mean": 0.03541659864668663, "eval_steps_per_second": 0.157, "step": 350 }, { "clip_ratio/high_max": 0.06420147884637117, "clip_ratio/high_mean": 0.06420147884637117, "clip_ratio/low_mean": 0.042410715483129025, "clip_ratio/low_min": 0.042410715483129025, "clip_ratio/region_mean": 0.1066121943295002, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 110.625, "completions/mean_terminated_length": 110.625, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.6541618332266808, "epoch": 0.01355421686746988, "frac_reward_zero_std": 0.0, "grad_norm": 7.404175281524658, "learning_rate": 8.93939393939394e-06, "loss": 0.0583, "num_tokens": 755121.0, "reward": 0.8215762376785278, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.8419463634490967, "reward_meter_std": 0.2106313854455948, "reward_std": 0.23777368664741516, "reward_total_composite_mean": 0.8215762376785278, "reward_total_composite_std": 0.23777370154857635, "reward_total_mean": 0.8215762376785278, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.8419463634490967, "rewards/meter/std": 0.2106313854455948, "rewards/total_composite/mean": 0.8215762376785278, "rewards/total_composite/std": 0.23777370154857635, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0185338258743286, "sampling/importance_sampling_ratio/min": 0.17011570930480957, "sampling/sampling_logp_difference/max": 1.7712764739990234, "sampling/sampling_logp_difference/mean": 0.10544459521770477, "step": 351 }, { "clip_ratio/high_max": 0.040051594376564026, "clip_ratio/high_mean": 0.040051594376564026, "clip_ratio/low_mean": 0.043291399255394936, "clip_ratio/low_min": 0.043291399255394936, "clip_ratio/region_mean": 0.08334299363195896, "completions/clipped_ratio": 0.0, "completions/max_length": 350.0, "completions/max_terminated_length": 350.0, "completions/mean_length": 312.0, "completions/mean_terminated_length": 312.0, "completions/min_length": 266.0, "completions/min_terminated_length": 266.0, "entropy": 0.5112155750393867, "epoch": 0.013592832869941304, "frac_reward_zero_std": 0.0, "grad_norm": 4.749355792999268, "learning_rate": 8.936363636363638e-06, "loss": 0.0229, "num_tokens": 759273.0, "reward": 0.3508853614330292, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8977272510528564, "reward_count_adherence_std": 0.0758657306432724, "reward_meter_mean": 0.6065003275871277, "reward_meter_std": 0.3428960144519806, "reward_std": 0.3484891951084137, "reward_total_composite_mean": 0.3508853614330292, "reward_total_composite_std": 0.3484891951084137, "reward_total_mean": 0.3508853614330292, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8977272510528564, "rewards/count_adherence/std": 0.0758657306432724, "rewards/meter/mean": 0.6065003275871277, "rewards/meter/std": 0.3428960144519806, "rewards/total_composite/mean": 0.3508853614330292, "rewards/total_composite/std": 0.3484891951084137, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0068981647491455, "sampling/importance_sampling_ratio/min": 0.048666857182979584, "sampling/sampling_logp_difference/max": 3.022757053375244, "sampling/sampling_logp_difference/mean": 0.09127286821603775, "step": 352 }, { "clip_ratio/high_max": 0.051137037575244904, "clip_ratio/high_mean": 0.051137037575244904, "clip_ratio/low_mean": 0.10074498318135738, "clip_ratio/low_min": 0.10074498318135738, "clip_ratio/region_mean": 0.1518820207566023, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 53.375, "completions/mean_terminated_length": 53.375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.9390326142311096, "epoch": 0.013631448872412728, "frac_reward_zero_std": 0.0, "grad_norm": 15.203680992126465, "learning_rate": 8.933333333333333e-06, "loss": 0.0551, "num_tokens": 760972.0, "reward": 0.17881934344768524, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.17881934344768524, "reward_meter_std": 0.21953943371772766, "reward_std": 0.21953943371772766, "reward_total_composite_mean": 0.17881934344768524, "reward_total_composite_std": 0.21953943371772766, "reward_total_mean": 0.17881934344768524, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.17881934344768524, "rewards/meter/std": 0.21953943371772766, "rewards/total_composite/mean": 0.17881934344768524, "rewards/total_composite/std": 0.21953943371772766, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006723403930664, "sampling/importance_sampling_ratio/min": 0.06635808944702148, "sampling/sampling_logp_difference/max": 2.7126896381378174, "sampling/sampling_logp_difference/mean": 0.16142068803310394, "step": 353 }, { "clip_ratio/high_max": 0.03562006680294871, "clip_ratio/high_mean": 0.03562006680294871, "clip_ratio/low_mean": 0.04170922236517072, "clip_ratio/low_min": 0.04170922236517072, "clip_ratio/region_mean": 0.07732928916811943, "completions/clipped_ratio": 0.0, "completions/max_length": 255.0, "completions/max_terminated_length": 255.0, "completions/mean_length": 217.5, "completions/mean_terminated_length": 217.5, "completions/min_length": 183.0, "completions/min_terminated_length": 183.0, "entropy": 0.4405891187489033, "epoch": 0.013670064874884152, "frac_reward_zero_std": 0.0, "grad_norm": 5.69448184967041, "learning_rate": 8.930303030303032e-06, "loss": 0.0918, "num_tokens": 764184.0, "reward": 0.43276864290237427, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9285714626312256, "reward_count_adherence_std": 0.07636035233736038, "reward_meter_mean": 0.47910094261169434, "reward_meter_std": 0.35389965772628784, "reward_std": 0.353183388710022, "reward_total_composite_mean": 0.43276864290237427, "reward_total_composite_std": 0.35318341851234436, "reward_total_mean": 0.43276864290237427, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9285714626312256, "rewards/count_adherence/std": 0.07636035233736038, "rewards/meter/mean": 0.47910094261169434, "rewards/meter/std": 0.35389965772628784, "rewards/total_composite/mean": 0.43276864290237427, "rewards/total_composite/std": 0.35318341851234436, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9991462230682373, "sampling/importance_sampling_ratio/min": 0.01798236183822155, "sampling/sampling_logp_difference/max": 4.018363952636719, "sampling/sampling_logp_difference/mean": 0.09170925617218018, "step": 354 }, { "clip_ratio/high_max": 0.0662611536681652, "clip_ratio/high_mean": 0.0662611536681652, "clip_ratio/low_mean": 0.013718276750296354, "clip_ratio/low_min": 0.013718276750296354, "clip_ratio/region_mean": 0.07997943041846156, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 62.75, "completions/mean_terminated_length": 62.75, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.6785079278051853, "epoch": 0.013708680877355576, "frac_reward_zero_std": 0.0, "grad_norm": 8.740988731384277, "learning_rate": 8.927272727272728e-06, "loss": 0.1883, "num_tokens": 765870.0, "reward": 0.7710193395614624, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.8165422677993774, "reward_meter_std": 0.2722550332546234, "reward_std": 0.3160322606563568, "reward_total_composite_mean": 0.7710193395614624, "reward_total_composite_std": 0.3160322606563568, "reward_total_mean": 0.7710193395614624, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.8165422677993774, "rewards/meter/std": 0.2722550332546234, "rewards/total_composite/mean": 0.7710193395614624, "rewards/total_composite/std": 0.3160322606563568, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9960544109344482, "sampling/importance_sampling_ratio/min": 0.2889011800289154, "sampling/sampling_logp_difference/max": 1.2416706085205078, "sampling/sampling_logp_difference/mean": 0.09419586509466171, "step": 355 }, { "clip_ratio/high_max": 0.029830908868461847, "clip_ratio/high_mean": 0.029830908868461847, "clip_ratio/low_mean": 0.02032768283970654, "clip_ratio/low_min": 0.02032768283970654, "clip_ratio/region_mean": 0.05015859170816839, "completions/clipped_ratio": 0.0, "completions/max_length": 174.0, "completions/max_terminated_length": 174.0, "completions/mean_length": 91.0, "completions/mean_terminated_length": 91.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.4440796282142401, "epoch": 0.013747296879827, "frac_reward_zero_std": 0.0, "grad_norm": 5.667863368988037, "learning_rate": 8.924242424242425e-06, "loss": 0.3162, "num_tokens": 767806.0, "reward": 0.5287183523178101, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.4432026445865631, "reward_meter_mean": 0.7748513221740723, "reward_meter_std": 0.3971291780471802, "reward_std": 0.421138733625412, "reward_total_composite_mean": 0.5287183523178101, "reward_total_composite_std": 0.4211387634277344, "reward_total_mean": 0.5287183523178101, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.4432026445865631, "rewards/meter/mean": 0.7748513221740723, "rewards/meter/std": 0.3971291780471802, "rewards/total_composite/mean": 0.5287183523178101, "rewards/total_composite/std": 0.4211387634277344, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.00663161277771, "sampling/importance_sampling_ratio/min": 0.207846999168396, "sampling/sampling_logp_difference/max": 1.5709530115127563, "sampling/sampling_logp_difference/mean": 0.057105425745248795, "step": 356 }, { "clip_ratio/high_max": 0.07131410390138626, "clip_ratio/high_mean": 0.07131410390138626, "clip_ratio/low_mean": 0.05875496123917401, "clip_ratio/low_min": 0.05875496123917401, "clip_ratio/region_mean": 0.13006906514056027, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 29.625, "completions/mean_terminated_length": 29.625, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 1.4358574822545052, "epoch": 0.013785912882298424, "frac_reward_zero_std": 0.0, "grad_norm": 14.779327392578125, "learning_rate": 8.921212121212122e-06, "loss": -0.0252, "num_tokens": 769331.0, "reward": 0.644232988357544, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.644232988357544, "reward_meter_std": 0.4240405559539795, "reward_std": 0.4240405559539795, "reward_total_composite_mean": 0.644232988357544, "reward_total_composite_std": 0.4240405559539795, "reward_total_mean": 0.644232988357544, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.644232988357544, "rewards/meter/std": 0.4240405559539795, "rewards/total_composite/mean": 0.644232988357544, "rewards/total_composite/std": 0.4240405559539795, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0188149213790894, "sampling/importance_sampling_ratio/min": 0.20378637313842773, "sampling/sampling_logp_difference/max": 1.5906829833984375, "sampling/sampling_logp_difference/mean": 0.12809233367443085, "step": 357 }, { "clip_ratio/high_max": 0.053119941614568233, "clip_ratio/high_mean": 0.053119941614568233, "clip_ratio/low_mean": 0.0018292682943865657, "clip_ratio/low_min": 0.0018292682943865657, "clip_ratio/region_mean": 0.0549492099089548, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 205.0, "completions/mean_length": 177.75, "completions/mean_terminated_length": 130.0, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.3757124850526452, "epoch": 0.013824528884769849, "frac_reward_zero_std": 0.0, "grad_norm": 2.133929491043091, "learning_rate": 8.91818181818182e-06, "loss": -0.0033, "num_tokens": 771649.0, "reward": 0.6629506349563599, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.34503278136253357, "reward_meter_mean": 0.7501189708709717, "reward_meter_std": 0.4528391361236572, "reward_std": 0.43371593952178955, "reward_total_composite_mean": 0.6629506349563599, "reward_total_composite_std": 0.43371596932411194, "reward_total_mean": 0.6629506349563599, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.34503278136253357, "rewards/meter/mean": 0.7501189708709717, "rewards/meter/std": 0.4528391361236572, "rewards/total_composite/mean": 0.6629506349563599, "rewards/total_composite/std": 0.43371596932411194, "sampling/importance_sampling_ratio/max": 1.9287148714065552, "sampling/importance_sampling_ratio/mean": 1.0138229131698608, "sampling/importance_sampling_ratio/min": 0.16741985082626343, "sampling/sampling_logp_difference/max": 1.7872505187988281, "sampling/sampling_logp_difference/mean": 0.053938575088977814, "step": 358 }, { "clip_ratio/high_max": 0.025988655164837837, "clip_ratio/high_mean": 0.025988655164837837, "clip_ratio/low_mean": 0.006502890028059483, "clip_ratio/low_min": 0.006502890028059483, "clip_ratio/region_mean": 0.03249154519289732, "completions/clipped_ratio": 0.0, "completions/max_length": 207.0, "completions/max_terminated_length": 207.0, "completions/mean_length": 189.25, "completions/mean_terminated_length": 189.25, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "entropy": 0.20762322936207056, "epoch": 0.013863144887241273, "frac_reward_zero_std": 0.0, "grad_norm": 4.576873779296875, "learning_rate": 8.915151515151515e-06, "loss": -0.0206, "num_tokens": 774739.0, "reward": 0.9521228075027466, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9521228075027466, "reward_meter_std": 0.1113143265247345, "reward_std": 0.1113143190741539, "reward_total_composite_mean": 0.9521228075027466, "reward_total_composite_std": 0.1113143265247345, "reward_total_mean": 0.9521228075027466, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9521228075027466, "rewards/meter/std": 0.1113143265247345, "rewards/total_composite/mean": 0.9521228075027466, "rewards/total_composite/std": 0.1113143265247345, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0031837224960327, "sampling/importance_sampling_ratio/min": 0.31054913997650146, "sampling/sampling_logp_difference/max": 1.1694130897521973, "sampling/sampling_logp_difference/mean": 0.033114705234766006, "step": 359 }, { "clip_ratio/high_max": 0.06353305862285197, "clip_ratio/high_mean": 0.06353305862285197, "clip_ratio/low_mean": 0.031017429195344448, "clip_ratio/low_min": 0.031017429195344448, "clip_ratio/region_mean": 0.09455048781819642, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 69.375, "completions/mean_terminated_length": 69.375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.9112853184342384, "epoch": 0.013901760889712697, "frac_reward_zero_std": 0.0, "grad_norm": 8.154752731323242, "learning_rate": 8.912121212121214e-06, "loss": 0.0254, "num_tokens": 776550.0, "reward": 0.7717971801757812, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7717971801757812, "reward_meter_std": 0.3203040659427643, "reward_std": 0.3203040659427643, "reward_total_composite_mean": 0.7717971801757812, "reward_total_composite_std": 0.3203040659427643, "reward_total_mean": 0.7717971801757812, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7717971801757812, "rewards/meter/std": 0.3203040659427643, "rewards/total_composite/mean": 0.7717971801757812, "rewards/total_composite/std": 0.3203040659427643, "sampling/importance_sampling_ratio/max": 1.804463267326355, "sampling/importance_sampling_ratio/mean": 1.0157387256622314, "sampling/importance_sampling_ratio/min": 0.2980525493621826, "sampling/sampling_logp_difference/max": 1.2104854583740234, "sampling/sampling_logp_difference/mean": 0.09423349797725677, "step": 360 }, { "clip_ratio/high_max": 0.07991194678470492, "clip_ratio/high_mean": 0.07991194678470492, "clip_ratio/low_mean": 0.026876877062022686, "clip_ratio/low_min": 0.026876877062022686, "clip_ratio/region_mean": 0.10678882384672761, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 97.125, "completions/mean_terminated_length": 97.125, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.8551923632621765, "epoch": 0.01394037689218412, "frac_reward_zero_std": 0.0, "grad_norm": 13.655211448669434, "learning_rate": 8.90909090909091e-06, "loss": 0.0533, "num_tokens": 778655.0, "reward": 0.9174784421920776, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9174784421920776, "reward_meter_std": 0.10890436172485352, "reward_std": 0.10890434682369232, "reward_total_composite_mean": 0.9174784421920776, "reward_total_composite_std": 0.10890436172485352, "reward_total_mean": 0.9174784421920776, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9174784421920776, "rewards/meter/std": 0.10890436172485352, "rewards/total_composite/mean": 0.9174784421920776, "rewards/total_composite/std": 0.10890436172485352, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0211211442947388, "sampling/importance_sampling_ratio/min": 0.16303564608097076, "sampling/sampling_logp_difference/max": 1.8137863874435425, "sampling/sampling_logp_difference/mean": 0.1321590393781662, "step": 361 }, { "clip_ratio/high_max": 0.021009880118072033, "clip_ratio/high_mean": 0.021009880118072033, "clip_ratio/low_mean": 0.004234739113599062, "clip_ratio/low_min": 0.004234739113599062, "clip_ratio/region_mean": 0.025244619231671095, "completions/clipped_ratio": 0.0, "completions/max_length": 141.0, "completions/max_terminated_length": 141.0, "completions/mean_length": 111.375, "completions/mean_terminated_length": 111.375, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.20824719313532114, "epoch": 0.013978992894655545, "frac_reward_zero_std": 0.0, "grad_norm": 3.6917600631713867, "learning_rate": 8.906060606060607e-06, "loss": 0.0839, "num_tokens": 781154.0, "reward": 0.9842470288276672, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9842470288276672, "reward_meter_std": 0.012328671291470528, "reward_std": 0.012328672222793102, "reward_total_composite_mean": 0.9842470288276672, "reward_total_composite_std": 0.012328671291470528, "reward_total_mean": 0.9842470288276672, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9842470288276672, "rewards/meter/std": 0.012328671291470528, "rewards/total_composite/mean": 0.9842470288276672, "rewards/total_composite/std": 0.012328671291470528, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0069833993911743, "sampling/importance_sampling_ratio/min": 0.2445802241563797, "sampling/sampling_logp_difference/max": 1.4082119464874268, "sampling/sampling_logp_difference/mean": 0.029706332832574844, "step": 362 }, { "clip_ratio/high_max": 0.07039879262447357, "clip_ratio/high_mean": 0.07039879262447357, "clip_ratio/low_mean": 0.030319053679704666, "clip_ratio/low_min": 0.030319053679704666, "clip_ratio/region_mean": 0.10071784630417824, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 70.625, "completions/mean_terminated_length": 70.625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.7273201942443848, "epoch": 0.014017608897126969, "frac_reward_zero_std": 0.0, "grad_norm": 10.716965675354004, "learning_rate": 8.903030303030304e-06, "loss": 0.0367, "num_tokens": 783007.0, "reward": 0.6237794160842896, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6237794160842896, "reward_meter_std": 0.3550654649734497, "reward_std": 0.3550654351711273, "reward_total_composite_mean": 0.6237794160842896, "reward_total_composite_std": 0.3550654649734497, "reward_total_mean": 0.6237794160842896, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6237794160842896, "rewards/meter/std": 0.3550654649734497, "rewards/total_composite/mean": 0.6237794160842896, "rewards/total_composite/std": 0.3550654649734497, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0215344429016113, "sampling/importance_sampling_ratio/min": 0.17098604142665863, "sampling/sampling_logp_difference/max": 1.7661733627319336, "sampling/sampling_logp_difference/mean": 0.10527131706476212, "step": 363 }, { "clip_ratio/high_max": 0.12814369797706604, "clip_ratio/high_mean": 0.12814369797706604, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/region_mean": 0.13707226980477571, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 55.375, "completions/mean_terminated_length": 55.375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 1.1656930446624756, "epoch": 0.014056224899598393, "frac_reward_zero_std": 0.0, "grad_norm": 10.589325904846191, "learning_rate": 8.900000000000001e-06, "loss": 0.0098, "num_tokens": 784698.0, "reward": 0.8658082485198975, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8658082485198975, "reward_meter_std": 0.2791600823402405, "reward_std": 0.2791600525379181, "reward_total_composite_mean": 0.8658082485198975, "reward_total_composite_std": 0.2791600823402405, "reward_total_mean": 0.8658082485198975, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8658082485198975, "rewards/meter/std": 0.2791600823402405, "rewards/total_composite/mean": 0.8658082485198975, "rewards/total_composite/std": 0.2791600823402405, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.013712763786316, "sampling/importance_sampling_ratio/min": 0.3191623091697693, "sampling/sampling_logp_difference/max": 1.1420555114746094, "sampling/sampling_logp_difference/mean": 0.13043774664402008, "step": 364 }, { "clip_ratio/high_max": 0.0560882892459631, "clip_ratio/high_mean": 0.0560882892459631, "clip_ratio/low_mean": 0.05067971581593156, "clip_ratio/low_min": 0.05067971581593156, "clip_ratio/region_mean": 0.10676800506189466, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 122.125, "completions/mean_terminated_length": 122.125, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.5795126520097256, "epoch": 0.014094840902069817, "frac_reward_zero_std": 0.0, "grad_norm": 13.111671447753906, "learning_rate": 8.896969696969697e-06, "loss": 0.1186, "num_tokens": 787243.0, "reward": 0.07550254464149475, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.07551360130310059, "reward_meter_std": 0.0628013163805008, "reward_std": 0.06281644105911255, "reward_total_composite_mean": 0.07550254464149475, "reward_total_composite_std": 0.06281644105911255, "reward_total_mean": 0.07550254464149475, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.07551360130310059, "rewards/meter/std": 0.0628013163805008, "rewards/total_composite/mean": 0.07550254464149475, "rewards/total_composite/std": 0.06281644105911255, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0089328289031982, "sampling/importance_sampling_ratio/min": 0.11512690037488937, "sampling/sampling_logp_difference/max": 2.1617202758789062, "sampling/sampling_logp_difference/mean": 0.13020947575569153, "step": 365 }, { "clip_ratio/high_max": 0.08675131388008595, "clip_ratio/high_mean": 0.08675131388008595, "clip_ratio/low_mean": 0.07220942713320255, "clip_ratio/low_min": 0.07220942713320255, "clip_ratio/region_mean": 0.1589607410132885, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 53.625, "completions/mean_terminated_length": 53.625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 2.051785424351692, "epoch": 0.014133456904541241, "frac_reward_zero_std": 0.0, "grad_norm": 13.758865356445312, "learning_rate": 8.893939393939394e-06, "loss": 0.0132, "num_tokens": 788896.0, "reward": 0.5461307764053345, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5461307764053345, "reward_meter_std": 0.36441224813461304, "reward_std": 0.36441221833229065, "reward_total_composite_mean": 0.5461307764053345, "reward_total_composite_std": 0.36441224813461304, "reward_total_mean": 0.5461307764053345, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5461307764053345, "rewards/meter/std": 0.36441224813461304, "rewards/total_composite/mean": 0.5461307764053345, "rewards/total_composite/std": 0.36441224813461304, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0263890027999878, "sampling/importance_sampling_ratio/min": 0.19857226312160492, "sampling/sampling_logp_difference/max": 1.6166021823883057, "sampling/sampling_logp_difference/mean": 0.1779501587152481, "step": 366 }, { "clip_ratio/high_max": 0.04118129098787904, "clip_ratio/high_mean": 0.04118129098787904, "clip_ratio/low_mean": 0.028543358203023672, "clip_ratio/low_min": 0.028543358203023672, "clip_ratio/region_mean": 0.06972464919090271, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 79.375, "completions/mean_terminated_length": 79.375, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.6636021360754967, "epoch": 0.014172072907012665, "frac_reward_zero_std": 0.0, "grad_norm": 9.988097190856934, "learning_rate": 8.890909090909091e-06, "loss": -0.0044, "num_tokens": 790811.0, "reward": 0.9885889291763306, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9885889291763306, "reward_meter_std": 0.0046616969630122185, "reward_std": 0.004661702550947666, "reward_total_composite_mean": 0.9885889291763306, "reward_total_composite_std": 0.0046616969630122185, "reward_total_mean": 0.9885889291763306, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9885889291763306, "rewards/meter/std": 0.0046616969630122185, "rewards/total_composite/mean": 0.9885889291763306, "rewards/total_composite/std": 0.0046616969630122185, "sampling/importance_sampling_ratio/max": 1.7074363231658936, "sampling/importance_sampling_ratio/mean": 1.0116159915924072, "sampling/importance_sampling_ratio/min": 0.2945983111858368, "sampling/sampling_logp_difference/max": 1.2221425771713257, "sampling/sampling_logp_difference/mean": 0.07036007940769196, "step": 367 }, { "clip_ratio/high_max": 0.06731301127001643, "clip_ratio/high_mean": 0.06731301127001643, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/region_mean": 0.07155029941350222, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 62.125, "completions/mean_terminated_length": 62.125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.4450516998767853, "epoch": 0.01421068890948409, "frac_reward_zero_std": 0.0, "grad_norm": 5.517111778259277, "learning_rate": 8.887878787878789e-06, "loss": -0.0084, "num_tokens": 792532.0, "reward": 0.8935509920120239, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8935509920120239, "reward_meter_std": 0.2326761782169342, "reward_std": 0.2326761931180954, "reward_total_composite_mean": 0.8935509920120239, "reward_total_composite_std": 0.2326761782169342, "reward_total_mean": 0.8935509920120239, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8935509920120239, "rewards/meter/std": 0.2326761782169342, "rewards/total_composite/mean": 0.8935509920120239, "rewards/total_composite/std": 0.2326761782169342, "sampling/importance_sampling_ratio/max": 1.846786618232727, "sampling/importance_sampling_ratio/mean": 0.9973874688148499, "sampling/importance_sampling_ratio/min": 0.2832942605018616, "sampling/sampling_logp_difference/max": 1.2612690925598145, "sampling/sampling_logp_difference/mean": 0.06450241804122925, "step": 368 }, { "clip_ratio/high_max": 0.051239289343357086, "clip_ratio/high_mean": 0.051239289343357086, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/region_mean": 0.057098664343357086, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.4530269633978605, "epoch": 0.014249304911955514, "frac_reward_zero_std": 0.0, "grad_norm": 4.8590497970581055, "learning_rate": 8.884848484848486e-06, "loss": -0.0431, "num_tokens": 794480.0, "reward": 0.9953228235244751, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953228235244751, "reward_meter_std": 0.008204305544495583, "reward_std": 0.008204314857721329, "reward_total_composite_mean": 0.9953228235244751, "reward_total_composite_std": 0.008204305544495583, "reward_total_mean": 0.9953228235244751, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953228235244751, "rewards/meter/std": 0.008204305544495583, "rewards/total_composite/mean": 0.9953228235244751, "rewards/total_composite/std": 0.008204305544495583, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0053033828735352, "sampling/importance_sampling_ratio/min": 0.23970215022563934, "sampling/sampling_logp_difference/max": 2.3124313354492188, "sampling/sampling_logp_difference/mean": 0.057661134749650955, "step": 369 }, { "clip_ratio/high_max": 0.007220303174108267, "clip_ratio/high_mean": 0.007220303174108267, "clip_ratio/low_mean": 0.023962138686329126, "clip_ratio/low_min": 0.023962138686329126, "clip_ratio/region_mean": 0.031182441860437393, "completions/clipped_ratio": 0.0, "completions/max_length": 193.0, "completions/max_terminated_length": 193.0, "completions/mean_length": 158.0, "completions/mean_terminated_length": 158.0, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.33616957906633615, "epoch": 0.014287920914426938, "frac_reward_zero_std": 0.0, "grad_norm": 3.5815742015838623, "learning_rate": 8.881818181818183e-06, "loss": -0.0679, "num_tokens": 797320.0, "reward": 0.8628234267234802, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_meter_mean": 0.9859833717346191, "reward_meter_std": 0.00827515497803688, "reward_std": 0.07765334099531174, "reward_total_composite_mean": 0.8628234267234802, "reward_total_composite_std": 0.07765333354473114, "reward_total_mean": 0.8628234267234802, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/meter/mean": 0.9859833717346191, "rewards/meter/std": 0.00827515497803688, "rewards/total_composite/mean": 0.8628234267234802, "rewards/total_composite/std": 0.07765333354473114, "sampling/importance_sampling_ratio/max": 1.8284692764282227, "sampling/importance_sampling_ratio/mean": 1.002156376838684, "sampling/importance_sampling_ratio/min": 0.2905518114566803, "sampling/sampling_logp_difference/max": 1.2359733581542969, "sampling/sampling_logp_difference/mean": 0.036213189363479614, "step": 370 }, { "clip_ratio/high_max": 0.02175675635226071, "clip_ratio/high_mean": 0.02175675635226071, "clip_ratio/low_mean": 0.02037237980403006, "clip_ratio/low_min": 0.02037237980403006, "clip_ratio/region_mean": 0.04212913615629077, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 63.125, "completions/mean_terminated_length": 63.125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.3037525750696659, "epoch": 0.014326536916898362, "frac_reward_zero_std": 0.0, "grad_norm": 4.787428379058838, "learning_rate": 8.87878787878788e-06, "loss": -0.0611, "num_tokens": 799041.0, "reward": 0.9925068020820618, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9925068020820618, "reward_meter_std": 0.004724609199911356, "reward_std": 0.004724607802927494, "reward_total_composite_mean": 0.9925068020820618, "reward_total_composite_std": 0.004724609199911356, "reward_total_mean": 0.9925068020820618, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9925068020820618, "rewards/meter/std": 0.004724609199911356, "rewards/total_composite/mean": 0.9925068020820618, "rewards/total_composite/std": 0.004724609199911356, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0087881088256836, "sampling/importance_sampling_ratio/min": 0.4250510334968567, "sampling/sampling_logp_difference/max": 1.3125829696655273, "sampling/sampling_logp_difference/mean": 0.03738410025835037, "step": 371 }, { "clip_ratio/high_max": 0.05502570327371359, "clip_ratio/high_mean": 0.05502570327371359, "clip_ratio/low_mean": 0.03239734377712011, "clip_ratio/low_min": 0.03239734377712011, "clip_ratio/region_mean": 0.0874230470508337, "completions/clipped_ratio": 0.0, "completions/max_length": 49.0, "completions/max_terminated_length": 49.0, "completions/mean_length": 47.125, "completions/mean_terminated_length": 47.125, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.453409057110548, "epoch": 0.014365152919369786, "frac_reward_zero_std": 0.0, "grad_norm": 10.848628044128418, "learning_rate": 8.875757575757576e-06, "loss": 0.0082, "num_tokens": 800618.0, "reward": 0.7270175218582153, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7270175218582153, "reward_meter_std": 0.33816584944725037, "reward_std": 0.33816584944725037, "reward_total_composite_mean": 0.7270175218582153, "reward_total_composite_std": 0.33816584944725037, "reward_total_mean": 0.7270175218582153, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7270175218582153, "rewards/meter/std": 0.33816584944725037, "rewards/total_composite/mean": 0.7270175218582153, "rewards/total_composite/std": 0.33816584944725037, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007200002670288, "sampling/importance_sampling_ratio/min": 0.22253604233264923, "sampling/sampling_logp_difference/max": 1.5026662349700928, "sampling/sampling_logp_difference/mean": 0.11081711202859879, "step": 372 }, { "clip_ratio/high_max": 0.08649335522204638, "clip_ratio/high_mean": 0.08649335522204638, "clip_ratio/low_mean": 0.03359880484640598, "clip_ratio/low_min": 0.03359880484640598, "clip_ratio/region_mean": 0.12009216006845236, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 51.625, "completions/mean_terminated_length": 51.625, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 1.7377965189516544, "epoch": 0.014403768921841212, "frac_reward_zero_std": 0.0, "grad_norm": 15.01648235321045, "learning_rate": 8.872727272727275e-06, "loss": -0.0215, "num_tokens": 802191.0, "reward": 0.8347938060760498, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8347938060760498, "reward_meter_std": 0.2780701220035553, "reward_std": 0.2780701220035553, "reward_total_composite_mean": 0.8347938060760498, "reward_total_composite_std": 0.2780701220035553, "reward_total_mean": 0.8347938060760498, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8347938060760498, "rewards/meter/std": 0.2780701220035553, "rewards/total_composite/mean": 0.8347938060760498, "rewards/total_composite/std": 0.2780701220035553, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0421912670135498, "sampling/importance_sampling_ratio/min": 0.2581785023212433, "sampling/sampling_logp_difference/max": 1.3541040420532227, "sampling/sampling_logp_difference/mean": 0.15455831587314606, "step": 373 }, { "clip_ratio/high_max": 0.10351844411343336, "clip_ratio/high_mean": 0.10351844411343336, "clip_ratio/low_mean": 0.026901003904640675, "clip_ratio/low_min": 0.026901003904640675, "clip_ratio/region_mean": 0.13041944801807404, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 77.375, "completions/mean_terminated_length": 77.375, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.7330310307443142, "epoch": 0.014442384924312636, "frac_reward_zero_std": 0.0, "grad_norm": 9.044235229492188, "learning_rate": 8.86969696969697e-06, "loss": 0.0391, "num_tokens": 804026.0, "reward": 0.8928734064102173, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8928734064102173, "reward_meter_std": 0.21326282620429993, "reward_std": 0.21326281130313873, "reward_total_composite_mean": 0.8928734064102173, "reward_total_composite_std": 0.21326282620429993, "reward_total_mean": 0.8928734064102173, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8928734064102173, "rewards/meter/std": 0.21326282620429993, "rewards/total_composite/mean": 0.8928734064102173, "rewards/total_composite/std": 0.21326282620429993, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0097415447235107, "sampling/importance_sampling_ratio/min": 0.20891344547271729, "sampling/sampling_logp_difference/max": 1.5658352375030518, "sampling/sampling_logp_difference/mean": 0.10972950607538223, "step": 374 }, { "clip_ratio/high_max": 0.051100376760587096, "clip_ratio/high_mean": 0.051100376760587096, "clip_ratio/low_mean": 0.009168443502858281, "clip_ratio/low_min": 0.009168443502858281, "clip_ratio/region_mean": 0.06026882026344538, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.7069630771875381, "epoch": 0.01448100092678406, "frac_reward_zero_std": 0.0, "grad_norm": 9.385233879089355, "learning_rate": 8.866666666666668e-06, "loss": 0.0183, "num_tokens": 805811.0, "reward": 0.9862740635871887, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9862740635871887, "reward_meter_std": 0.022809647023677826, "reward_std": 0.02280966006219387, "reward_total_composite_mean": 0.9862740635871887, "reward_total_composite_std": 0.022809647023677826, "reward_total_mean": 0.9862740635871887, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9862740635871887, "rewards/meter/std": 0.022809647023677826, "rewards/total_composite/mean": 0.9862740635871887, "rewards/total_composite/std": 0.022809647023677826, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005605697631836, "sampling/importance_sampling_ratio/min": 0.24791260063648224, "sampling/sampling_logp_difference/max": 1.394679069519043, "sampling/sampling_logp_difference/mean": 0.08005134761333466, "step": 375 }, { "clip_ratio/high_max": 0.049564928747713566, "clip_ratio/high_mean": 0.049564928747713566, "clip_ratio/low_mean": 0.07232486410066485, "clip_ratio/low_min": 0.07232486410066485, "clip_ratio/region_mean": 0.12188979284837842, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 86.75, "completions/mean_terminated_length": 86.75, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 1.4634561464190483, "epoch": 0.014519616929255484, "frac_reward_zero_std": 0.0, "grad_norm": 7.467469215393066, "learning_rate": 8.863636363636365e-06, "loss": 0.0159, "num_tokens": 807817.0, "reward": 0.38385850191116333, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.38385850191116333, "reward_meter_std": 0.3889022171497345, "reward_std": 0.3889022171497345, "reward_total_composite_mean": 0.38385850191116333, "reward_total_composite_std": 0.3889022171497345, "reward_total_mean": 0.38385850191116333, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.38385850191116333, "rewards/meter/std": 0.3889022171497345, "rewards/total_composite/mean": 0.38385850191116333, "rewards/total_composite/std": 0.3889022171497345, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0250762701034546, "sampling/importance_sampling_ratio/min": 0.2584901452064514, "sampling/sampling_logp_difference/max": 1.3528976440429688, "sampling/sampling_logp_difference/mean": 0.14060647785663605, "step": 376 }, { "clip_ratio/high_max": 0.028778444975614548, "clip_ratio/high_mean": 0.028778444975614548, "clip_ratio/low_mean": 0.004890488460659981, "clip_ratio/low_min": 0.004890488460659981, "clip_ratio/region_mean": 0.03366893343627453, "completions/clipped_ratio": 0.0, "completions/max_length": 195.0, "completions/max_terminated_length": 195.0, "completions/mean_length": 181.25, "completions/mean_terminated_length": 181.25, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.25179606676101685, "epoch": 0.014558232931726908, "frac_reward_zero_std": 0.0, "grad_norm": 2.5315823554992676, "learning_rate": 8.860606060606062e-06, "loss": -0.0199, "num_tokens": 810779.0, "reward": 0.7281081080436707, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.10681165754795074, "reward_meter_mean": 0.8494396209716797, "reward_meter_std": 0.3451205790042877, "reward_std": 0.314047247171402, "reward_total_composite_mean": 0.7281081080436707, "reward_total_composite_std": 0.31404727697372437, "reward_total_mean": 0.7281081080436707, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.10681165754795074, "rewards/meter/mean": 0.8494396209716797, "rewards/meter/std": 0.3451205790042877, "rewards/total_composite/mean": 0.7281081080436707, "rewards/total_composite/std": 0.31404727697372437, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015827417373657, "sampling/importance_sampling_ratio/min": 0.2383430302143097, "sampling/sampling_logp_difference/max": 1.434044361114502, "sampling/sampling_logp_difference/mean": 0.04079859331250191, "step": 377 }, { "clip_ratio/high_max": 0.0347325021866709, "clip_ratio/high_mean": 0.0347325021866709, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/region_mean": 0.043661074014380574, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 143.0, "completions/mean_terminated_length": 90.28572082519531, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.33806798979640007, "epoch": 0.014596848934198332, "frac_reward_zero_std": 0.0, "grad_norm": 6.6282148361206055, "learning_rate": 8.857575757575758e-06, "loss": -0.1719, "num_tokens": 812651.0, "reward": 0.8677093982696533, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.24800792336463928, "reward_meter_mean": 0.9918551445007324, "reward_meter_std": 0.004135291092097759, "reward_std": 0.24586959183216095, "reward_total_composite_mean": 0.8677093982696533, "reward_total_composite_std": 0.24586959183216095, "reward_total_mean": 0.8677093982696533, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.24800792336463928, "rewards/meter/mean": 0.9918551445007324, "rewards/meter/std": 0.004135291092097759, "rewards/total_composite/mean": 0.8677093982696533, "rewards/total_composite/std": 0.24586959183216095, "sampling/importance_sampling_ratio/max": 1.980355143547058, "sampling/importance_sampling_ratio/mean": 1.008686900138855, "sampling/importance_sampling_ratio/min": 0.29936423897743225, "sampling/sampling_logp_difference/max": 1.2060942649841309, "sampling/sampling_logp_difference/mean": 0.052827268838882446, "step": 378 }, { "clip_ratio/high_max": 0.08018230739980936, "clip_ratio/high_mean": 0.08018230739980936, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.08018230739980936, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 130.625, "completions/mean_terminated_length": 76.14286041259766, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.7669309824705124, "epoch": 0.014635464936669756, "frac_reward_zero_std": 0.0, "grad_norm": 1.1005628108978271, "learning_rate": 8.854545454545455e-06, "loss": -0.1795, "num_tokens": 814504.0, "reward": 0.9303869605064392, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9917647838592529, "reward_meter_std": 0.009205368347465992, "reward_std": 0.17772506177425385, "reward_total_composite_mean": 0.9303869605064392, "reward_total_composite_std": 0.17772506177425385, "reward_total_mean": 0.9303869605064392, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9917647838592529, "rewards/meter/std": 0.009205368347465992, "rewards/total_composite/mean": 0.9303869605064392, "rewards/total_composite/std": 0.17772506177425385, "sampling/importance_sampling_ratio/max": 1.736937165260315, "sampling/importance_sampling_ratio/mean": 1.0020660161972046, "sampling/importance_sampling_ratio/min": 0.3145061433315277, "sampling/sampling_logp_difference/max": 1.1567516326904297, "sampling/sampling_logp_difference/mean": 0.08945769816637039, "step": 379 }, { "clip_ratio/high_max": 0.07530082948505878, "clip_ratio/high_mean": 0.07530082948505878, "clip_ratio/low_mean": 0.05674390681087971, "clip_ratio/low_min": 0.05674390681087971, "clip_ratio/region_mean": 0.1320447362959385, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 62.0, "completions/mean_terminated_length": 62.0, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 1.2989665120840073, "epoch": 0.01467408093914118, "frac_reward_zero_std": 0.0, "grad_norm": 9.77833366394043, "learning_rate": 8.851515151515152e-06, "loss": -0.0121, "num_tokens": 816336.0, "reward": 0.8162009716033936, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8162009716033936, "reward_meter_std": 0.21702320873737335, "reward_std": 0.21702319383621216, "reward_total_composite_mean": 0.8162009716033936, "reward_total_composite_std": 0.21702320873737335, "reward_total_mean": 0.8162009716033936, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8162009716033936, "rewards/meter/std": 0.21702320873737335, "rewards/total_composite/mean": 0.8162009716033936, "rewards/total_composite/std": 0.21702320873737335, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0254578590393066, "sampling/importance_sampling_ratio/min": 0.2928999066352844, "sampling/sampling_logp_difference/max": 1.2279243469238281, "sampling/sampling_logp_difference/mean": 0.13395410776138306, "step": 380 }, { "clip_ratio/high_max": 0.02148950519040227, "clip_ratio/high_mean": 0.02148950519040227, "clip_ratio/low_mean": 0.02306468691676855, "clip_ratio/low_min": 0.02306468691676855, "clip_ratio/region_mean": 0.04455419210717082, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 77.375, "completions/mean_terminated_length": 77.375, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.47446247562766075, "epoch": 0.014712696941612605, "frac_reward_zero_std": 0.0, "grad_norm": 5.939858913421631, "learning_rate": 8.84848484848485e-06, "loss": 0.0106, "num_tokens": 818123.0, "reward": 0.9950414299964905, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9950414299964905, "reward_meter_std": 0.005652880761772394, "reward_std": 0.005652868654578924, "reward_total_composite_mean": 0.9950414299964905, "reward_total_composite_std": 0.005652880761772394, "reward_total_mean": 0.9950414299964905, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9950414299964905, "rewards/meter/std": 0.005652880761772394, "rewards/total_composite/mean": 0.9950414299964905, "rewards/total_composite/std": 0.005652880761772394, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0126311779022217, "sampling/importance_sampling_ratio/min": 0.18580889701843262, "sampling/sampling_logp_difference/max": 1.6830365657806396, "sampling/sampling_logp_difference/mean": 0.05509962514042854, "step": 381 }, { "clip_ratio/high_max": 0.023234429769217968, "clip_ratio/high_mean": 0.023234429769217968, "clip_ratio/low_mean": 0.0074404762126505375, "clip_ratio/low_min": 0.0074404762126505375, "clip_ratio/region_mean": 0.030674905981868505, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 185.0, "completions/mean_length": 211.75, "completions/mean_terminated_length": 168.85714721679688, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.311492882668972, "epoch": 0.014751312944084029, "frac_reward_zero_std": 0.0, "grad_norm": 1.8641679286956787, "learning_rate": 8.845454545454547e-06, "loss": -0.2242, "num_tokens": 820713.0, "reward": 0.7125877141952515, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.23299294710159302, "reward_meter_mean": 0.9491947889328003, "reward_meter_std": 0.10156966745853424, "reward_std": 0.2403973639011383, "reward_total_composite_mean": 0.7125877141952515, "reward_total_composite_std": 0.2403973489999771, "reward_total_mean": 0.7125877141952515, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.23299294710159302, "rewards/meter/mean": 0.9491947889328003, "rewards/meter/std": 0.10156966745853424, "rewards/total_composite/mean": 0.7125877141952515, "rewards/total_composite/std": 0.2403973489999771, "sampling/importance_sampling_ratio/max": 1.9695637226104736, "sampling/importance_sampling_ratio/mean": 1.0095936059951782, "sampling/importance_sampling_ratio/min": 0.24216197431087494, "sampling/sampling_logp_difference/max": 1.4181485176086426, "sampling/sampling_logp_difference/mean": 0.04226767271757126, "step": 382 }, { "clip_ratio/high_max": 0.01344697322929278, "clip_ratio/high_mean": 0.01344697322929278, "clip_ratio/low_mean": 0.0015822785208001733, "clip_ratio/low_min": 0.0015822785208001733, "clip_ratio/region_mean": 0.015029251750092953, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 252.0, "completions/mean_length": 263.0, "completions/mean_terminated_length": 227.4285888671875, "completions/min_length": 199.0, "completions/min_terminated_length": 199.0, "entropy": 0.10115705709904432, "epoch": 0.014789928946555453, "frac_reward_zero_std": 0.0, "grad_norm": 1.574141263961792, "learning_rate": 8.842424242424244e-06, "loss": -0.1634, "num_tokens": 823881.0, "reward": 0.5856432914733887, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.2314550280570984, "reward_meter_mean": 0.8664563298225403, "reward_meter_std": 0.2947894036769867, "reward_std": 0.2903149425983429, "reward_total_composite_mean": 0.5856432914733887, "reward_total_composite_std": 0.2903149425983429, "reward_total_mean": 0.5856432914733887, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/meter/mean": 0.8664563298225403, "rewards/meter/std": 0.2947894036769867, "rewards/total_composite/mean": 0.5856432914733887, "rewards/total_composite/std": 0.2903149425983429, "sampling/importance_sampling_ratio/max": 1.8257579803466797, "sampling/importance_sampling_ratio/mean": 1.000448226928711, "sampling/importance_sampling_ratio/min": 0.3171911835670471, "sampling/sampling_logp_difference/max": 1.1482505798339844, "sampling/sampling_logp_difference/mean": 0.01734515093266964, "step": 383 }, { "clip_ratio/high_max": 0.03935463528614491, "clip_ratio/high_mean": 0.03935463528614491, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/region_mean": 0.04791627905797213, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 132.125, "completions/mean_terminated_length": 77.85714721679688, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.3592350035905838, "epoch": 0.014828544949026877, "frac_reward_zero_std": 0.0, "grad_norm": 2.7595345973968506, "learning_rate": 8.83939393939394e-06, "loss": -0.0853, "num_tokens": 825706.0, "reward": 0.8233171105384827, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.8857226967811584, "reward_meter_std": 0.2998782992362976, "reward_std": 0.32403311133384705, "reward_total_composite_mean": 0.8233171105384827, "reward_total_composite_std": 0.32403311133384705, "reward_total_mean": 0.8233171105384827, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.8857226967811584, "rewards/meter/std": 0.2998782992362976, "rewards/total_composite/mean": 0.8233171105384827, "rewards/total_composite/std": 0.32403311133384705, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005861759185791, "sampling/importance_sampling_ratio/min": 0.19981907308101654, "sampling/sampling_logp_difference/max": 1.6103429794311523, "sampling/sampling_logp_difference/mean": 0.05464348942041397, "step": 384 }, { "clip_ratio/high_max": 0.041496834717690945, "clip_ratio/high_mean": 0.041496834717690945, "clip_ratio/low_mean": 0.003689236124046147, "clip_ratio/low_min": 0.003689236124046147, "clip_ratio/region_mean": 0.04518607084173709, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 178.75, "completions/mean_terminated_length": 67.66667175292969, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.43436889722943306, "epoch": 0.014867160951498301, "frac_reward_zero_std": 0.0, "grad_norm": 2.582751512527466, "learning_rate": 8.836363636363637e-06, "loss": -0.0509, "num_tokens": 827248.0, "reward": 0.714938223361969, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.714938223361969, "reward_meter_std": 0.36748242378234863, "reward_std": 0.36748242378234863, "reward_total_composite_mean": 0.714938223361969, "reward_total_composite_std": 0.36748242378234863, "reward_total_mean": 0.714938223361969, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.714938223361969, "rewards/meter/std": 0.36748242378234863, "rewards/total_composite/mean": 0.714938223361969, "rewards/total_composite/std": 0.36748242378234863, "sampling/importance_sampling_ratio/max": 1.566494107246399, "sampling/importance_sampling_ratio/mean": 1.0119549036026, "sampling/importance_sampling_ratio/min": 0.33843517303466797, "sampling/sampling_logp_difference/max": 1.0834226608276367, "sampling/sampling_logp_difference/mean": 0.07517319172620773, "step": 385 }, { "clip_ratio/high_max": 0.07191554573364556, "clip_ratio/high_mean": 0.07191554573364556, "clip_ratio/low_mean": 0.03804087173193693, "clip_ratio/low_min": 0.03804087173193693, "clip_ratio/region_mean": 0.10995641746558249, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 58.25, "completions/mean_terminated_length": 58.25, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 1.1986883580684662, "epoch": 0.014905776953969725, "frac_reward_zero_std": 0.0, "grad_norm": 13.948408126831055, "learning_rate": 8.833333333333334e-06, "loss": 0.0492, "num_tokens": 828954.0, "reward": 0.690985918045044, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.690985918045044, "reward_meter_std": 0.42123687267303467, "reward_std": 0.42123687267303467, "reward_total_composite_mean": 0.690985918045044, "reward_total_composite_std": 0.42123687267303467, "reward_total_mean": 0.690985918045044, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.690985918045044, "rewards/meter/std": 0.42123687267303467, "rewards/total_composite/mean": 0.690985918045044, "rewards/total_composite/std": 0.42123687267303467, "sampling/importance_sampling_ratio/max": 1.8247418403625488, "sampling/importance_sampling_ratio/mean": 1.0141727924346924, "sampling/importance_sampling_ratio/min": 0.1829366534948349, "sampling/sampling_logp_difference/max": 1.698615312576294, "sampling/sampling_logp_difference/mean": 0.12392318993806839, "step": 386 }, { "clip_ratio/high_max": 0.007820297265425324, "clip_ratio/high_mean": 0.007820297265425324, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007820297265425324, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 315.25, "completions/mean_terminated_length": 287.14288330078125, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.06537747196853161, "epoch": 0.01494439295644115, "frac_reward_zero_std": 0.0, "grad_norm": 0.3322877585887909, "learning_rate": 8.830303030303031e-06, "loss": -0.2826, "num_tokens": 832724.0, "reward": 0.6194590330123901, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2121320217847824, "reward_meter_mean": 0.9916845560073853, "reward_meter_std": 0.01541493646800518, "reward_std": 0.21031685173511505, "reward_total_composite_mean": 0.6194590330123901, "reward_total_composite_std": 0.21031685173511505, "reward_total_mean": 0.6194590330123901, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2121320217847824, "rewards/meter/mean": 0.9916845560073853, "rewards/meter/std": 0.01541493646800518, "rewards/total_composite/mean": 0.6194590330123901, "rewards/total_composite/std": 0.21031685173511505, "sampling/importance_sampling_ratio/max": 1.4588537216186523, "sampling/importance_sampling_ratio/mean": 1.0010322332382202, "sampling/importance_sampling_ratio/min": 0.15520599484443665, "sampling/sampling_logp_difference/max": 1.863002061843872, "sampling/sampling_logp_difference/mean": 0.011599292047321796, "step": 387 }, { "clip_ratio/high_max": 0.01489788806065917, "clip_ratio/high_mean": 0.01489788806065917, "clip_ratio/low_mean": 0.00827294704504311, "clip_ratio/low_min": 0.00827294704504311, "clip_ratio/region_mean": 0.02317083510570228, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 92.875, "completions/mean_terminated_length": 92.875, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.24626289308071136, "epoch": 0.014983008958912573, "frac_reward_zero_std": 0.0, "grad_norm": 3.59706449508667, "learning_rate": 8.827272727272727e-06, "loss": 0.0025, "num_tokens": 834907.0, "reward": 0.9105833172798157, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.9513072967529297, "reward_meter_std": 0.04710450395941734, "reward_std": 0.11427939683198929, "reward_total_composite_mean": 0.9105833172798157, "reward_total_composite_std": 0.11427941173315048, "reward_total_mean": 0.9105833172798157, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.9513072967529297, "rewards/meter/std": 0.04710450395941734, "rewards/total_composite/mean": 0.9105833172798157, "rewards/total_composite/std": 0.11427941173315048, "sampling/importance_sampling_ratio/max": 1.8268436193466187, "sampling/importance_sampling_ratio/mean": 1.0046768188476562, "sampling/importance_sampling_ratio/min": 0.3529474139213562, "sampling/sampling_logp_difference/max": 1.0414361953735352, "sampling/sampling_logp_difference/mean": 0.02844572812318802, "step": 388 }, { "clip_ratio/high_max": 0.09026568429544568, "clip_ratio/high_mean": 0.09026568429544568, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/region_mean": 0.09702244121581316, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 34.25, "completions/mean_terminated_length": 34.25, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.777605514973402, "epoch": 0.015021624961383997, "frac_reward_zero_std": 0.0, "grad_norm": 10.221504211425781, "learning_rate": 8.824242424242426e-06, "loss": 0.0368, "num_tokens": 836445.0, "reward": 0.8630082011222839, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8630082011222839, "reward_meter_std": 0.34829336404800415, "reward_std": 0.34829336404800415, "reward_total_composite_mean": 0.8630082011222839, "reward_total_composite_std": 0.34829336404800415, "reward_total_mean": 0.8630082011222839, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8630082011222839, "rewards/meter/std": 0.34829336404800415, "rewards/total_composite/mean": 0.8630082011222839, "rewards/total_composite/std": 0.34829336404800415, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0091283321380615, "sampling/importance_sampling_ratio/min": 0.10067791491746902, "sampling/sampling_logp_difference/max": 2.2958288192749023, "sampling/sampling_logp_difference/mean": 0.11211102455854416, "step": 389 }, { "clip_ratio/high_max": 0.07300483155995607, "clip_ratio/high_mean": 0.07300483155995607, "clip_ratio/low_mean": 0.030689293053001165, "clip_ratio/low_min": 0.030689293053001165, "clip_ratio/region_mean": 0.10369412461295724, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 54.75, "completions/mean_terminated_length": 54.75, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 1.3469589613378048, "epoch": 0.015060240963855422, "frac_reward_zero_std": 0.0, "grad_norm": 8.775293350219727, "learning_rate": 8.821212121212121e-06, "loss": -0.0172, "num_tokens": 838331.0, "reward": 0.48685652017593384, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.48685652017593384, "reward_meter_std": 0.3812946379184723, "reward_std": 0.3812946081161499, "reward_total_composite_mean": 0.48685652017593384, "reward_total_composite_std": 0.3812946379184723, "reward_total_mean": 0.48685652017593384, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.48685652017593384, "rewards/meter/std": 0.3812946379184723, "rewards/total_composite/mean": 0.48685652017593384, "rewards/total_composite/std": 0.3812946379184723, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.029207468032837, "sampling/importance_sampling_ratio/min": 0.2566058933734894, "sampling/sampling_logp_difference/max": 1.3602138757705688, "sampling/sampling_logp_difference/mean": 0.12760156393051147, "step": 390 }, { "clip_ratio/high_max": 0.0027604734350461513, "clip_ratio/high_mean": 0.0027604734350461513, "clip_ratio/low_mean": 0.003784051747061312, "clip_ratio/low_min": 0.003784051747061312, "clip_ratio/region_mean": 0.0065445251821074635, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 342.875, "completions/mean_terminated_length": 318.71429443359375, "completions/min_length": 295.0, "completions/min_terminated_length": 295.0, "entropy": 0.051004831213504076, "epoch": 0.015098856966326846, "frac_reward_zero_std": 0.0, "grad_norm": 0.7848347425460815, "learning_rate": 8.818181818181819e-06, "loss": -0.1899, "num_tokens": 842434.0, "reward": 0.5062733888626099, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6136363744735718, "reward_count_adherence_std": 0.25132471323013306, "reward_meter_mean": 0.7168420553207397, "reward_meter_std": 0.3972436785697937, "reward_std": 0.2856999337673187, "reward_total_composite_mean": 0.5062733888626099, "reward_total_composite_std": 0.2856999337673187, "reward_total_mean": 0.5062733888626099, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6136363744735718, "rewards/count_adherence/std": 0.25132471323013306, "rewards/meter/mean": 0.7168420553207397, "rewards/meter/std": 0.3972436785697937, "rewards/total_composite/mean": 0.5062733888626099, "rewards/total_composite/std": 0.2856999337673187, "sampling/importance_sampling_ratio/max": 1.6002336740493774, "sampling/importance_sampling_ratio/mean": 1.0016242265701294, "sampling/importance_sampling_ratio/min": 0.3409028947353363, "sampling/sampling_logp_difference/max": 1.076157569885254, "sampling/sampling_logp_difference/mean": 0.008826361037790775, "step": 391 }, { "clip_ratio/high_max": 0.007834467338398099, "clip_ratio/high_mean": 0.007834467338398099, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007834467338398099, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 373.25, "completions/mean_terminated_length": 290.0, "completions/min_length": 283.0, "completions/min_terminated_length": 283.0, "entropy": 0.05020246421918273, "epoch": 0.01513747296879827, "frac_reward_zero_std": 0.0, "grad_norm": 0.6587200164794922, "learning_rate": 8.815151515151516e-06, "loss": -0.3491, "num_tokens": 845556.0, "reward": 0.5244123935699463, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5277777910232544, "reward_count_adherence_std": 0.37912923097610474, "reward_meter_mean": 0.8696821331977844, "reward_meter_std": 0.35143736004829407, "reward_std": 0.3767135441303253, "reward_total_composite_mean": 0.5244123935699463, "reward_total_composite_std": 0.3767135441303253, "reward_total_mean": 0.5244123935699463, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5277777910232544, "rewards/count_adherence/std": 0.37912923097610474, "rewards/meter/mean": 0.8696821331977844, "rewards/meter/std": 0.35143736004829407, "rewards/total_composite/mean": 0.5244123935699463, "rewards/total_composite/std": 0.3767135441303253, "sampling/importance_sampling_ratio/max": 1.8669248819351196, "sampling/importance_sampling_ratio/mean": 1.002270221710205, "sampling/importance_sampling_ratio/min": 0.33716630935668945, "sampling/sampling_logp_difference/max": 1.0871790647506714, "sampling/sampling_logp_difference/mean": 0.015206166543066502, "step": 392 }, { "clip_ratio/high_max": 0.06486599263735116, "clip_ratio/high_mean": 0.06486599263735116, "clip_ratio/low_mean": 0.010939412750303745, "clip_ratio/low_min": 0.010939412750303745, "clip_ratio/region_mean": 0.0758054053876549, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 76.375, "completions/mean_terminated_length": 76.375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.4674673527479172, "epoch": 0.015176088971269694, "frac_reward_zero_std": 0.0, "grad_norm": 8.689268112182617, "learning_rate": 8.812121212121213e-06, "loss": 0.0322, "num_tokens": 847311.0, "reward": 0.8588340878486633, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8588340878486633, "reward_meter_std": 0.2888956665992737, "reward_std": 0.2888956665992737, "reward_total_composite_mean": 0.8588340878486633, "reward_total_composite_std": 0.2888956665992737, "reward_total_mean": 0.8588340878486633, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8588340878486633, "rewards/meter/std": 0.2888956665992737, "rewards/total_composite/mean": 0.8588340878486633, "rewards/total_composite/std": 0.2888956665992737, "sampling/importance_sampling_ratio/max": 1.9332025051116943, "sampling/importance_sampling_ratio/mean": 0.9985421895980835, "sampling/importance_sampling_ratio/min": 0.13957762718200684, "sampling/sampling_logp_difference/max": 1.9691343307495117, "sampling/sampling_logp_difference/mean": 0.07139719277620316, "step": 393 }, { "clip_ratio/high_max": 0.028151679784059525, "clip_ratio/high_mean": 0.028151679784059525, "clip_ratio/low_mean": 0.007181095774285495, "clip_ratio/low_min": 0.007181095774285495, "clip_ratio/region_mean": 0.03533277555834502, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.37163497880101204, "epoch": 0.015214704973741118, "frac_reward_zero_std": 0.0, "grad_norm": 4.890065670013428, "learning_rate": 8.809090909090909e-06, "loss": 0.0681, "num_tokens": 849152.0, "reward": 0.9758315086364746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9758315086364746, "reward_meter_std": 0.012030374258756638, "reward_std": 0.012030377984046936, "reward_total_composite_mean": 0.9758315086364746, "reward_total_composite_std": 0.012030374258756638, "reward_total_mean": 0.9758315086364746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9758315086364746, "rewards/meter/std": 0.012030374258756638, "rewards/total_composite/mean": 0.9758315086364746, "rewards/total_composite/std": 0.012030374258756638, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004264235496521, "sampling/importance_sampling_ratio/min": 0.4598883390426636, "sampling/sampling_logp_difference/max": 0.8468852043151855, "sampling/sampling_logp_difference/mean": 0.048741716891527176, "step": 394 }, { "clip_ratio/high_max": 0.012422295869328082, "clip_ratio/high_mean": 0.012422295869328082, "clip_ratio/low_mean": 0.009408602491021156, "clip_ratio/low_min": 0.009408602491021156, "clip_ratio/region_mean": 0.021830898360349238, "completions/clipped_ratio": 0.0, "completions/max_length": 265.0, "completions/max_terminated_length": 265.0, "completions/mean_length": 228.625, "completions/mean_terminated_length": 228.625, "completions/min_length": 186.0, "completions/min_terminated_length": 186.0, "entropy": 0.17339316615834832, "epoch": 0.015253320976212542, "frac_reward_zero_std": 0.0, "grad_norm": 2.795933485031128, "learning_rate": 8.806060606060608e-06, "loss": -0.0587, "num_tokens": 852613.0, "reward": 0.8257385492324829, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_meter_mean": 0.9906454086303711, "reward_meter_std": 0.012306920252740383, "reward_std": 0.3412370979785919, "reward_total_composite_mean": 0.8257385492324829, "reward_total_composite_std": 0.3412371277809143, "reward_total_mean": 0.8257385492324829, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/meter/mean": 0.9906454086303711, "rewards/meter/std": 0.012306920252740383, "rewards/total_composite/mean": 0.8257385492324829, "rewards/total_composite/std": 0.3412371277809143, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015069246292114, "sampling/importance_sampling_ratio/min": 0.3412272036075592, "sampling/sampling_logp_difference/max": 1.5854034423828125, "sampling/sampling_logp_difference/mean": 0.02395275980234146, "step": 395 }, { "clip_ratio/high_max": 0.034834470599889755, "clip_ratio/high_mean": 0.034834470599889755, "clip_ratio/low_mean": 0.03667238587513566, "clip_ratio/low_min": 0.03667238587513566, "clip_ratio/region_mean": 0.07150685647502542, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.7259577997028828, "epoch": 0.015291936978683966, "frac_reward_zero_std": 0.0, "grad_norm": 15.071035385131836, "learning_rate": 8.803030303030303e-06, "loss": -0.0516, "num_tokens": 854021.0, "reward": 0.9902995824813843, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9902995824813843, "reward_meter_std": 0.008138173259794712, "reward_std": 0.008138181641697884, "reward_total_composite_mean": 0.9902995824813843, "reward_total_composite_std": 0.008138173259794712, "reward_total_mean": 0.9902995824813843, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9902995824813843, "rewards/meter/std": 0.008138173259794712, "rewards/total_composite/mean": 0.9902995824813843, "rewards/total_composite/std": 0.008138173259794712, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0173804759979248, "sampling/importance_sampling_ratio/min": 0.2798110842704773, "sampling/sampling_logp_difference/max": 1.2736406326293945, "sampling/sampling_logp_difference/mean": 0.08739858120679855, "step": 396 }, { "clip_ratio/high_max": 0.04474749346263707, "clip_ratio/high_mean": 0.04474749346263707, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/region_mean": 0.056111130164936185, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 134.25, "completions/mean_terminated_length": 80.28572082519531, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.4520561061799526, "epoch": 0.01533055298115539, "frac_reward_zero_std": 0.0, "grad_norm": 4.2064313888549805, "learning_rate": 8.8e-06, "loss": 0.0085, "num_tokens": 855791.0, "reward": 0.865929901599884, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.865929901599884, "reward_meter_std": 0.3411285877227783, "reward_std": 0.3411285877227783, "reward_total_composite_mean": 0.865929901599884, "reward_total_composite_std": 0.3411285877227783, "reward_total_mean": 0.865929901599884, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.865929901599884, "rewards/meter/std": 0.3411285877227783, "rewards/total_composite/mean": 0.865929901599884, "rewards/total_composite/std": 0.3411285877227783, "sampling/importance_sampling_ratio/max": 1.667251706123352, "sampling/importance_sampling_ratio/mean": 1.0096566677093506, "sampling/importance_sampling_ratio/min": 0.19086399674415588, "sampling/sampling_logp_difference/max": 1.6561942100524902, "sampling/sampling_logp_difference/mean": 0.07304941862821579, "step": 397 }, { "clip_ratio/high_max": 0.04711233067791909, "clip_ratio/high_mean": 0.04711233067791909, "clip_ratio/low_mean": 0.005040322430431843, "clip_ratio/low_min": 0.005040322430431843, "clip_ratio/region_mean": 0.05215265310835093, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 89.875, "completions/mean_terminated_length": 89.875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.38309746980667114, "epoch": 0.015369168983626814, "frac_reward_zero_std": 0.0, "grad_norm": 4.518268585205078, "learning_rate": 8.796969696969698e-06, "loss": 0.1437, "num_tokens": 857766.0, "reward": 0.9344407320022583, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9967172145843506, "reward_meter_std": 0.003067870857194066, "reward_std": 0.17628978192806244, "reward_total_composite_mean": 0.9344407320022583, "reward_total_composite_std": 0.17628978192806244, "reward_total_mean": 0.9344407320022583, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9967172145843506, "rewards/meter/std": 0.003067870857194066, "rewards/total_composite/mean": 0.9344407320022583, "rewards/total_composite/std": 0.17628978192806244, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002355694770813, "sampling/importance_sampling_ratio/min": 0.2568609118461609, "sampling/sampling_logp_difference/max": 1.3592205047607422, "sampling/sampling_logp_difference/mean": 0.05183165520429611, "step": 398 }, { "clip_ratio/high_max": 0.03413120610639453, "clip_ratio/high_mean": 0.03413120610639453, "clip_ratio/low_mean": 0.03256993810646236, "clip_ratio/low_min": 0.03256993810646236, "clip_ratio/region_mean": 0.06670114421285689, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 53.75, "completions/mean_terminated_length": 53.75, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.45951317623257637, "epoch": 0.015407784986098239, "frac_reward_zero_std": 0.0, "grad_norm": 10.965712547302246, "learning_rate": 8.793939393939395e-06, "loss": 0.0428, "num_tokens": 859500.0, "reward": 0.26512396335601807, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.37250423431396484, "reward_meter_std": 0.42634570598602295, "reward_std": 0.39319032430648804, "reward_total_composite_mean": 0.26512396335601807, "reward_total_composite_std": 0.3931903541088104, "reward_total_mean": 0.26512396335601807, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.37250423431396484, "rewards/meter/std": 0.42634570598602295, "rewards/total_composite/mean": 0.26512396335601807, "rewards/total_composite/std": 0.3931903541088104, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001451015472412, "sampling/importance_sampling_ratio/min": 0.08536264300346375, "sampling/sampling_logp_difference/max": 2.4608466625213623, "sampling/sampling_logp_difference/mean": 0.0987580195069313, "step": 399 }, { "clip_ratio/high_max": 0.011500443739350885, "clip_ratio/high_mean": 0.011500443739350885, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.011500443739350885, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 157.0, "completions/mean_length": 196.25, "completions/mean_terminated_length": 151.1428680419922, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.0809963047504425, "epoch": 0.015446400988569663, "frac_reward_zero_std": 0.0, "grad_norm": 0.6731557250022888, "learning_rate": 8.790909090909092e-06, "loss": -0.2368, "num_tokens": 862030.0, "reward": 0.6456302404403687, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.65625, "reward_count_adherence_std": 0.2651650309562683, "reward_meter_mean": 0.9850339889526367, "reward_meter_std": 0.02169320359826088, "reward_std": 0.2613680958747864, "reward_total_composite_mean": 0.6456302404403687, "reward_total_composite_std": 0.26136812567710876, "reward_total_mean": 0.6456302404403687, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.65625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/meter/mean": 0.9850339889526367, "rewards/meter/std": 0.02169320359826088, "rewards/total_composite/mean": 0.6456302404403687, "rewards/total_composite/std": 0.26136812567710876, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002915382385254, "sampling/importance_sampling_ratio/min": 0.18668344616889954, "sampling/sampling_logp_difference/max": 1.6783409118652344, "sampling/sampling_logp_difference/mean": 0.020483043044805527, "step": 400 }, { "epoch": 0.015446400988569663, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.41346153846153844, "eval_completions/max_length": 512.0, "eval_completions/max_terminated_length": 400.53846153846155, "eval_completions/mean_length": 344.08653846153845, "eval_completions/mean_terminated_length": 226.2978057861328, "eval_completions/min_length": 85.23076923076923, "eval_completions/min_terminated_length": 85.23076923076923, "eval_entropy": 0.19937350027836287, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 862030.0, "eval_reward": 0.4161693981060615, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.6621637413134942, "eval_reward_count_adherence_std": 0.3351786213998611, "eval_reward_meter_mean": 0.6171861015833341, "eval_reward_meter_std": 0.42123945630513704, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.4161693981060615, "eval_reward_total_composite_std": 0.378701776266098, "eval_reward_total_mean": 0.4161693981060615, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.6621637413134942, "eval_rewards/count_adherence/std": 0.3351786213998611, "eval_rewards/meter/mean": 0.6171861015833341, "eval_rewards/meter/std": 0.42123945630513704, "eval_rewards/total_composite/mean": 0.4161693981060615, "eval_rewards/total_composite/std": 0.378701776266098, "eval_runtime": 96.6201, "eval_samples_per_second": 1.076, "eval_sampling/importance_sampling_ratio/max": 1.3838598086283758, "eval_sampling/importance_sampling_ratio/mean": 1.0045312092854426, "eval_sampling/importance_sampling_ratio/min": 0.40342745643395644, "eval_sampling/sampling_logp_difference/max": 0.9377481570610633, "eval_sampling/sampling_logp_difference/mean": 0.018011176170637973, "eval_steps_per_second": 0.135, "step": 400 }, { "clip_ratio/high_max": 0.011423650896176696, "clip_ratio/high_mean": 0.011423650896176696, "clip_ratio/low_mean": 0.04286885913461447, "clip_ratio/low_min": 0.04286885913461447, "clip_ratio/region_mean": 0.05429251003079116, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 70.375, "completions/mean_terminated_length": 70.375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.6498333364725113, "epoch": 0.015485016991041087, "frac_reward_zero_std": 0.0, "grad_norm": 6.882446765899658, "learning_rate": 8.787878787878788e-06, "loss": 0.0529, "num_tokens": 863793.0, "reward": 0.5120395421981812, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5120395421981812, "reward_meter_std": 0.39659610390663147, "reward_std": 0.39659610390663147, "reward_total_composite_mean": 0.5120395421981812, "reward_total_composite_std": 0.39659610390663147, "reward_total_mean": 0.5120395421981812, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5120395421981812, "rewards/meter/std": 0.39659610390663147, "rewards/total_composite/mean": 0.5120395421981812, "rewards/total_composite/std": 0.39659610390663147, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0185896158218384, "sampling/importance_sampling_ratio/min": 0.43671727180480957, "sampling/sampling_logp_difference/max": 0.8284692764282227, "sampling/sampling_logp_difference/mean": 0.079770028591156, "step": 401 }, { "clip_ratio/high_max": 0.01652618101797998, "clip_ratio/high_mean": 0.01652618101797998, "clip_ratio/low_mean": 0.013676032423973083, "clip_ratio/low_min": 0.013676032423973083, "clip_ratio/region_mean": 0.030202213441953063, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 90.875, "completions/mean_terminated_length": 90.875, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2551422342658043, "epoch": 0.01552363299351251, "frac_reward_zero_std": 0.0, "grad_norm": 3.843118190765381, "learning_rate": 8.784848484848487e-06, "loss": -0.0316, "num_tokens": 865768.0, "reward": 0.9842378497123718, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9842378497123718, "reward_meter_std": 0.0201681200414896, "reward_std": 0.020168133080005646, "reward_total_composite_mean": 0.9842378497123718, "reward_total_composite_std": 0.0201681200414896, "reward_total_mean": 0.9842378497123718, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9842378497123718, "rewards/meter/std": 0.0201681200414896, "rewards/total_composite/mean": 0.9842378497123718, "rewards/total_composite/std": 0.0201681200414896, "sampling/importance_sampling_ratio/max": 1.9674099683761597, "sampling/importance_sampling_ratio/mean": 1.0152220726013184, "sampling/importance_sampling_ratio/min": 0.31482240557670593, "sampling/sampling_logp_difference/max": 1.155746579170227, "sampling/sampling_logp_difference/mean": 0.03747996687889099, "step": 402 }, { "clip_ratio/high_max": 0.028843345178756863, "clip_ratio/high_mean": 0.028843345178756863, "clip_ratio/low_mean": 0.000936329597607255, "clip_ratio/low_min": 0.000936329597607255, "clip_ratio/region_mean": 0.029779674776364118, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 217.5, "completions/mean_terminated_length": 175.42857360839844, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.3300076089799404, "epoch": 0.015562248995983935, "frac_reward_zero_std": 0.0, "grad_norm": 1.8304016590118408, "learning_rate": 8.781818181818182e-06, "loss": -0.113, "num_tokens": 868356.0, "reward": 0.5801796913146973, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5833333730697632, "reward_count_adherence_std": 0.29546844959259033, "reward_meter_mean": 0.9523094892501831, "reward_meter_std": 0.1186181977391243, "reward_std": 0.29439738392829895, "reward_total_composite_mean": 0.5801796913146973, "reward_total_composite_std": 0.29439741373062134, "reward_total_mean": 0.5801796913146973, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5833333730697632, "rewards/count_adherence/std": 0.29546844959259033, "rewards/meter/mean": 0.9523094892501831, "rewards/meter/std": 0.1186181977391243, "rewards/total_composite/mean": 0.5801796913146973, "rewards/total_composite/std": 0.29439741373062134, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0064983367919922, "sampling/importance_sampling_ratio/min": 0.33814290165901184, "sampling/sampling_logp_difference/max": 1.0842866897583008, "sampling/sampling_logp_difference/mean": 0.043538108468055725, "step": 403 }, { "clip_ratio/high_max": 0.029422825085930526, "clip_ratio/high_mean": 0.029422825085930526, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/region_mean": 0.035104643437080085, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 187.25, "completions/mean_terminated_length": 79.0, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.26876694336533546, "epoch": 0.015600864998455359, "frac_reward_zero_std": 0.0, "grad_norm": 1.5644749402999878, "learning_rate": 8.77878787878788e-06, "loss": -0.1182, "num_tokens": 870166.0, "reward": 0.5707218647003174, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_meter_mean": 0.7447091341018677, "reward_meter_std": 0.3070685863494873, "reward_std": 0.45596015453338623, "reward_total_composite_mean": 0.5707218647003174, "reward_total_composite_std": 0.45596015453338623, "reward_total_mean": 0.5707218647003174, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/meter/mean": 0.7447091341018677, "rewards/meter/std": 0.3070685863494873, "rewards/total_composite/mean": 0.5707218647003174, "rewards/total_composite/std": 0.45596015453338623, "sampling/importance_sampling_ratio/max": 1.7909971475601196, "sampling/importance_sampling_ratio/mean": 1.0064294338226318, "sampling/importance_sampling_ratio/min": 0.2336532175540924, "sampling/sampling_logp_difference/max": 1.4539172649383545, "sampling/sampling_logp_difference/mean": 0.06043194606900215, "step": 404 }, { "clip_ratio/high_max": 0.0038860102649778128, "clip_ratio/high_mean": 0.0038860102649778128, "clip_ratio/low_mean": 0.012407735688611865, "clip_ratio/low_min": 0.012407735688611865, "clip_ratio/region_mean": 0.016293745953589678, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 256.0, "completions/mean_length": 300.75, "completions/mean_terminated_length": 230.33334350585938, "completions/min_length": 193.0, "completions/min_terminated_length": 193.0, "entropy": 0.12349590379744768, "epoch": 0.015639481000926783, "frac_reward_zero_std": 0.0, "grad_norm": 1.1297802925109863, "learning_rate": 8.775757575757577e-06, "loss": 0.181, "num_tokens": 872988.0, "reward": 0.23586036264896393, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6875, "reward_count_adherence_std": 0.22160132229328156, "reward_meter_mean": 0.37604400515556335, "reward_meter_std": 0.43697869777679443, "reward_std": 0.3007969558238983, "reward_total_composite_mean": 0.23586036264896393, "reward_total_composite_std": 0.3007969558238983, "reward_total_mean": 0.23586036264896393, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6875, "rewards/count_adherence/std": 0.22160132229328156, "rewards/meter/mean": 0.37604400515556335, "rewards/meter/std": 0.43697869777679443, "rewards/total_composite/mean": 0.23586036264896393, "rewards/total_composite/std": 0.3007969558238983, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0042189359664917, "sampling/importance_sampling_ratio/min": 0.2869817018508911, "sampling/sampling_logp_difference/max": 1.2483367919921875, "sampling/sampling_logp_difference/mean": 0.023270348086953163, "step": 405 }, { "clip_ratio/high_max": 0.0061827958561480045, "clip_ratio/high_mean": 0.0061827958561480045, "clip_ratio/low_mean": 0.011723854579031467, "clip_ratio/low_min": 0.011723854579031467, "clip_ratio/region_mean": 0.017906650435179472, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 62.625, "completions/mean_terminated_length": 62.625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.21097453031688929, "epoch": 0.015678097003398207, "frac_reward_zero_std": 0.0, "grad_norm": 4.1059088706970215, "learning_rate": 8.772727272727274e-06, "loss": 0.0324, "num_tokens": 874753.0, "reward": 0.9913897514343262, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9913897514343262, "reward_meter_std": 0.003956401254981756, "reward_std": 0.003956401254981756, "reward_total_composite_mean": 0.9913897514343262, "reward_total_composite_std": 0.003956401254981756, "reward_total_mean": 0.9913897514343262, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9913897514343262, "rewards/meter/std": 0.003956401254981756, "rewards/total_composite/mean": 0.9913897514343262, "rewards/total_composite/std": 0.003956401254981756, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0122790336608887, "sampling/importance_sampling_ratio/min": 0.4850950837135315, "sampling/sampling_logp_difference/max": 0.9925615787506104, "sampling/sampling_logp_difference/mean": 0.027037320658564568, "step": 406 }, { "clip_ratio/high_max": 0.016171834780834615, "clip_ratio/high_mean": 0.016171834780834615, "clip_ratio/low_mean": 0.011579734331462532, "clip_ratio/low_min": 0.011579734331462532, "clip_ratio/region_mean": 0.027751569112297148, "completions/clipped_ratio": 0.0, "completions/max_length": 168.0, "completions/max_terminated_length": 168.0, "completions/mean_length": 138.375, "completions/mean_terminated_length": 138.375, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.1530680824071169, "epoch": 0.01571671300586963, "frac_reward_zero_std": 0.0, "grad_norm": 3.022797107696533, "learning_rate": 8.76969696969697e-06, "loss": 0.0382, "num_tokens": 877196.0, "reward": 0.46666404604911804, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6999959945678711, "reward_meter_std": 0.3435265123844147, "reward_std": 0.22901766002178192, "reward_total_composite_mean": 0.46666404604911804, "reward_total_composite_std": 0.22901767492294312, "reward_total_mean": 0.46666404604911804, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6999959945678711, "rewards/meter/std": 0.3435265123844147, "rewards/total_composite/mean": 0.46666404604911804, "rewards/total_composite/std": 0.22901767492294312, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028660297393799, "sampling/importance_sampling_ratio/min": 0.24043114483356476, "sampling/sampling_logp_difference/max": 1.4253215789794922, "sampling/sampling_logp_difference/mean": 0.022317711263895035, "step": 407 }, { "clip_ratio/high_max": 0.08028767257928848, "clip_ratio/high_mean": 0.08028767257928848, "clip_ratio/low_mean": 0.036312646232545376, "clip_ratio/low_min": 0.036312646232545376, "clip_ratio/region_mean": 0.11660031881183386, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 59.0, "completions/mean_terminated_length": 59.0, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.9289319105446339, "epoch": 0.015755329008341055, "frac_reward_zero_std": 0.0, "grad_norm": 10.218791007995605, "learning_rate": 8.766666666666669e-06, "loss": 0.1037, "num_tokens": 878828.0, "reward": 0.6509081721305847, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6509081721305847, "reward_meter_std": 0.4415837228298187, "reward_std": 0.44158369302749634, "reward_total_composite_mean": 0.6509081721305847, "reward_total_composite_std": 0.4415837228298187, "reward_total_mean": 0.6509081721305847, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6509081721305847, "rewards/meter/std": 0.4415837228298187, "rewards/total_composite/mean": 0.6509081721305847, "rewards/total_composite/std": 0.4415837228298187, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.02310311794281, "sampling/importance_sampling_ratio/min": 0.06255759298801422, "sampling/sampling_logp_difference/max": 2.771667718887329, "sampling/sampling_logp_difference/mean": 0.11866805702447891, "step": 408 }, { "clip_ratio/high_max": 0.009341755299828947, "clip_ratio/high_mean": 0.009341755299828947, "clip_ratio/low_mean": 0.0006944444612599909, "clip_ratio/low_min": 0.0006944444612599909, "clip_ratio/region_mean": 0.010036199761088938, "completions/clipped_ratio": 0.0, "completions/max_length": 216.0, "completions/max_terminated_length": 216.0, "completions/mean_length": 175.625, "completions/mean_terminated_length": 175.625, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.07792735006660223, "epoch": 0.01579394501081248, "frac_reward_zero_std": 0.0, "grad_norm": 1.702805995941162, "learning_rate": 8.763636363636364e-06, "loss": 0.0206, "num_tokens": 881641.0, "reward": 0.7126193046569824, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9501590132713318, "reward_meter_std": 0.11174096912145615, "reward_std": 0.08380571752786636, "reward_total_composite_mean": 0.7126193046569824, "reward_total_composite_std": 0.08380572497844696, "reward_total_mean": 0.7126193046569824, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9501590132713318, "rewards/meter/std": 0.11174096912145615, "rewards/total_composite/mean": 0.7126193046569824, "rewards/total_composite/std": 0.08380572497844696, "sampling/importance_sampling_ratio/max": 1.7096338272094727, "sampling/importance_sampling_ratio/mean": 1.001296043395996, "sampling/importance_sampling_ratio/min": 0.38079074025154114, "sampling/sampling_logp_difference/max": 0.9655053615570068, "sampling/sampling_logp_difference/mean": 0.013060882687568665, "step": 409 }, { "clip_ratio/high_max": 0.012198542011901736, "clip_ratio/high_mean": 0.012198542011901736, "clip_ratio/low_mean": 0.003342245938256383, "clip_ratio/low_min": 0.003342245938256383, "clip_ratio/region_mean": 0.01554078795015812, "completions/clipped_ratio": 0.0, "completions/max_length": 244.0, "completions/max_terminated_length": 244.0, "completions/mean_length": 197.625, "completions/mean_terminated_length": 197.625, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "entropy": 0.0969978254288435, "epoch": 0.015832561013283904, "frac_reward_zero_std": 0.0, "grad_norm": 2.0530569553375244, "learning_rate": 8.760606060606061e-06, "loss": -0.0088, "num_tokens": 884542.0, "reward": 0.7310122847557068, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.8907101154327393, "reward_meter_std": 0.20771878957748413, "reward_std": 0.1586739867925644, "reward_total_composite_mean": 0.7310122847557068, "reward_total_composite_std": 0.1586739867925644, "reward_total_mean": 0.7310122847557068, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.8907101154327393, "rewards/meter/std": 0.20771878957748413, "rewards/total_composite/mean": 0.7310122847557068, "rewards/total_composite/std": 0.1586739867925644, "sampling/importance_sampling_ratio/max": 1.6074455976486206, "sampling/importance_sampling_ratio/mean": 0.9990739822387695, "sampling/importance_sampling_ratio/min": 0.009806273505091667, "sampling/sampling_logp_difference/max": 4.624732971191406, "sampling/sampling_logp_difference/mean": 0.017095215618610382, "step": 410 }, { "clip_ratio/high_max": 0.02881905622780323, "clip_ratio/high_mean": 0.02881905622780323, "clip_ratio/low_mean": 0.005724609247408807, "clip_ratio/low_min": 0.005724609247408807, "clip_ratio/region_mean": 0.03454366547521204, "completions/clipped_ratio": 0.0, "completions/max_length": 141.0, "completions/max_terminated_length": 141.0, "completions/mean_length": 105.375, "completions/mean_terminated_length": 105.375, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.23645233362913132, "epoch": 0.015871177015755328, "frac_reward_zero_std": 0.0, "grad_norm": 4.408254623413086, "learning_rate": 8.757575757575759e-06, "loss": 0.1971, "num_tokens": 886841.0, "reward": 0.8437220454216003, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_meter_mean": 0.9555299282073975, "reward_meter_std": 0.10529791563749313, "reward_std": 0.2139330804347992, "reward_total_composite_mean": 0.8437220454216003, "reward_total_composite_std": 0.2139330953359604, "reward_total_mean": 0.8437220454216003, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/meter/mean": 0.9555299282073975, "rewards/meter/std": 0.10529791563749313, "rewards/total_composite/mean": 0.8437220454216003, "rewards/total_composite/std": 0.2139330953359604, "sampling/importance_sampling_ratio/max": 1.7728590965270996, "sampling/importance_sampling_ratio/mean": 1.0038602352142334, "sampling/importance_sampling_ratio/min": 0.2763468325138092, "sampling/sampling_logp_difference/max": 1.2860984802246094, "sampling/sampling_logp_difference/mean": 0.03925402835011482, "step": 411 }, { "clip_ratio/high_max": 0.005883037491003051, "clip_ratio/high_mean": 0.005883037491003051, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005883037491003051, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 413.5, "completions/mean_terminated_length": 399.4285888671875, "completions/min_length": 378.0, "completions/min_terminated_length": 378.0, "entropy": 0.03046043962240219, "epoch": 0.015909793018226752, "frac_reward_zero_std": 0.0, "grad_norm": 0.3299441337585449, "learning_rate": 8.754545454545456e-06, "loss": -0.2932, "num_tokens": 891477.0, "reward": 0.8021494150161743, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8194444179534912, "reward_count_adherence_std": 0.29057809710502625, "reward_meter_mean": 0.8727073669433594, "reward_meter_std": 0.34727609157562256, "reward_std": 0.3275650143623352, "reward_total_composite_mean": 0.8021494150161743, "reward_total_composite_std": 0.3275650441646576, "reward_total_mean": 0.8021494150161743, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8194444179534912, "rewards/count_adherence/std": 0.29057809710502625, "rewards/meter/mean": 0.8727073669433594, "rewards/meter/std": 0.34727609157562256, "rewards/total_composite/mean": 0.8021494150161743, "rewards/total_composite/std": 0.3275650441646576, "sampling/importance_sampling_ratio/max": 1.6943782567977905, "sampling/importance_sampling_ratio/mean": 0.9996734857559204, "sampling/importance_sampling_ratio/min": 0.22663362324237823, "sampling/sampling_logp_difference/max": 1.4844205379486084, "sampling/sampling_logp_difference/mean": 0.007050658110529184, "step": 412 }, { "clip_ratio/high_max": 0.05882604233920574, "clip_ratio/high_mean": 0.05882604233920574, "clip_ratio/low_mean": 0.012824675533920527, "clip_ratio/low_min": 0.012824675533920527, "clip_ratio/region_mean": 0.07165071787312627, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 167.125, "completions/mean_terminated_length": 52.16666793823242, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.5243904180824757, "epoch": 0.015948409020698176, "frac_reward_zero_std": 0.0, "grad_norm": 2.0739166736602783, "learning_rate": 8.751515151515151e-06, "loss": 0.0018, "num_tokens": 892998.0, "reward": 0.5062292814254761, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_meter_mean": 0.5100301504135132, "reward_meter_std": 0.5075311064720154, "reward_std": 0.5117626786231995, "reward_total_composite_mean": 0.5062292814254761, "reward_total_composite_std": 0.5117626786231995, "reward_total_mean": 0.5062292814254761, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/meter/mean": 0.5100301504135132, "rewards/meter/std": 0.5075311064720154, "rewards/total_composite/mean": 0.5062292814254761, "rewards/total_composite/std": 0.5117626786231995, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01053786277771, "sampling/importance_sampling_ratio/min": 0.5460512638092041, "sampling/sampling_logp_difference/max": 0.7841591835021973, "sampling/sampling_logp_difference/mean": 0.06872761994600296, "step": 413 }, { "clip_ratio/high_max": 0.06984005169942975, "clip_ratio/high_mean": 0.06984005169942975, "clip_ratio/low_mean": 0.009469697251915932, "clip_ratio/low_min": 0.009469697251915932, "clip_ratio/region_mean": 0.07930974895134568, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 57.625, "completions/mean_terminated_length": 57.625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.582982636988163, "epoch": 0.0159870250231696, "frac_reward_zero_std": 0.0, "grad_norm": 10.703254699707031, "learning_rate": 8.748484848484849e-06, "loss": 0.0655, "num_tokens": 894651.0, "reward": 0.9551189541816711, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9551189541816711, "reward_meter_std": 0.09812162071466446, "reward_std": 0.09812159836292267, "reward_total_composite_mean": 0.9551189541816711, "reward_total_composite_std": 0.09812162071466446, "reward_total_mean": 0.9551189541816711, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9551189541816711, "rewards/meter/std": 0.09812162071466446, "rewards/total_composite/mean": 0.9551189541816711, "rewards/total_composite/std": 0.09812162071466446, "sampling/importance_sampling_ratio/max": 1.675656795501709, "sampling/importance_sampling_ratio/mean": 1.0104643106460571, "sampling/importance_sampling_ratio/min": 0.28781425952911377, "sampling/sampling_logp_difference/max": 1.2454400062561035, "sampling/sampling_logp_difference/mean": 0.07875344157218933, "step": 414 }, { "clip_ratio/high_max": 0.04313966212794185, "clip_ratio/high_mean": 0.04313966212794185, "clip_ratio/low_mean": 0.03434053680393845, "clip_ratio/low_min": 0.03434053680393845, "clip_ratio/region_mean": 0.0774801989318803, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 92.0, "completions/mean_length": 112.125, "completions/mean_terminated_length": 55.000003814697266, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.6476233620196581, "epoch": 0.016025641025641024, "frac_reward_zero_std": 0.0, "grad_norm": 4.649420261383057, "learning_rate": 8.745454545454546e-06, "loss": 0.0492, "num_tokens": 896156.0, "reward": 0.28754958510398865, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_meter_mean": 0.2886698246002197, "reward_meter_std": 0.40278124809265137, "reward_std": 0.40368756651878357, "reward_total_composite_mean": 0.28754958510398865, "reward_total_composite_std": 0.40368756651878357, "reward_total_mean": 0.28754958510398865, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/meter/mean": 0.2886698246002197, "rewards/meter/std": 0.40278124809265137, "rewards/total_composite/mean": 0.28754958510398865, "rewards/total_composite/std": 0.40368756651878357, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0061570405960083, "sampling/importance_sampling_ratio/min": 0.2186972051858902, "sampling/sampling_logp_difference/max": 1.5200672149658203, "sampling/sampling_logp_difference/mean": 0.10733786970376968, "step": 415 }, { "clip_ratio/high_max": 0.062005657935515046, "clip_ratio/high_mean": 0.062005657935515046, "clip_ratio/low_mean": 0.03538359794765711, "clip_ratio/low_min": 0.03538359794765711, "clip_ratio/region_mean": 0.09738925588317215, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 57.875, "completions/mean_terminated_length": 57.875, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.6767406091094017, "epoch": 0.01606425702811245, "frac_reward_zero_std": 0.0, "grad_norm": 7.304360866546631, "learning_rate": 8.742424242424243e-06, "loss": 0.0358, "num_tokens": 897867.0, "reward": 0.9717522859573364, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9717522859573364, "reward_meter_std": 0.027185985818505287, "reward_std": 0.027185987681150436, "reward_total_composite_mean": 0.9717522859573364, "reward_total_composite_std": 0.027185985818505287, "reward_total_mean": 0.9717522859573364, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9717522859573364, "rewards/meter/std": 0.027185985818505287, "rewards/total_composite/mean": 0.9717522859573364, "rewards/total_composite/std": 0.027185985818505287, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0156903266906738, "sampling/importance_sampling_ratio/min": 0.19506938755512238, "sampling/sampling_logp_difference/max": 1.6343998908996582, "sampling/sampling_logp_difference/mean": 0.09120234102010727, "step": 416 }, { "clip_ratio/high_max": 0.021424076927360147, "clip_ratio/high_mean": 0.021424076927360147, "clip_ratio/low_mean": 0.002363372186664492, "clip_ratio/low_min": 0.002363372186664492, "clip_ratio/region_mean": 0.02378744911402464, "completions/clipped_ratio": 0.0, "completions/max_length": 344.0, "completions/max_terminated_length": 344.0, "completions/mean_length": 227.125, "completions/mean_terminated_length": 227.125, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.2837667972780764, "epoch": 0.016102873030583872, "frac_reward_zero_std": 0.0, "grad_norm": 1.8169100284576416, "learning_rate": 8.73939393939394e-06, "loss": 0.1247, "num_tokens": 901164.0, "reward": 0.5552949905395508, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.14880475401878357, "reward_meter_mean": 0.6192874312400818, "reward_meter_std": 0.45972275733947754, "reward_std": 0.43653252720832825, "reward_total_composite_mean": 0.5552949905395508, "reward_total_composite_std": 0.43653252720832825, "reward_total_mean": 0.5552949905395508, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.14880475401878357, "rewards/meter/mean": 0.6192874312400818, "rewards/meter/std": 0.45972275733947754, "rewards/total_composite/mean": 0.5552949905395508, "rewards/total_composite/std": 0.43653252720832825, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0012174844741821, "sampling/importance_sampling_ratio/min": 0.27433061599731445, "sampling/sampling_logp_difference/max": 1.3905742168426514, "sampling/sampling_logp_difference/mean": 0.02308918908238411, "step": 417 }, { "clip_ratio/high_max": 0.03858941700309515, "clip_ratio/high_mean": 0.03858941700309515, "clip_ratio/low_mean": 0.11292503494769335, "clip_ratio/low_min": 0.11292503494769335, "clip_ratio/region_mean": 0.1515144519507885, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 50.125, "completions/mean_terminated_length": 50.125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.8871809840202332, "epoch": 0.016141489033055297, "frac_reward_zero_std": 0.0, "grad_norm": 21.298921585083008, "learning_rate": 8.736363636363638e-06, "loss": -0.0317, "num_tokens": 902909.0, "reward": 0.32849666476249695, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.32849666476249695, "reward_meter_std": 0.32036224007606506, "reward_std": 0.32036224007606506, "reward_total_composite_mean": 0.32849666476249695, "reward_total_composite_std": 0.32036224007606506, "reward_total_mean": 0.32849666476249695, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.32849666476249695, "rewards/meter/std": 0.32036224007606506, "rewards/total_composite/mean": 0.32849666476249695, "rewards/total_composite/std": 0.32036224007606506, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0049675703048706, "sampling/importance_sampling_ratio/min": 0.14722536504268646, "sampling/sampling_logp_difference/max": 2.284748077392578, "sampling/sampling_logp_difference/mean": 0.13514763116836548, "step": 418 }, { "clip_ratio/high_max": 0.05910295504145324, "clip_ratio/high_mean": 0.05910295504145324, "clip_ratio/low_mean": 0.00957037159241736, "clip_ratio/low_min": 0.00957037159241736, "clip_ratio/region_mean": 0.0686733266338706, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 73.75, "completions/mean_terminated_length": 73.75, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.5439852885901928, "epoch": 0.01618010503552672, "frac_reward_zero_std": 0.0, "grad_norm": 6.602462291717529, "learning_rate": 8.733333333333333e-06, "loss": 0.0702, "num_tokens": 904739.0, "reward": 0.799231767654419, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.799231767654419, "reward_meter_std": 0.3458506464958191, "reward_std": 0.3458506166934967, "reward_total_composite_mean": 0.799231767654419, "reward_total_composite_std": 0.3458506464958191, "reward_total_mean": 0.799231767654419, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.799231767654419, "rewards/meter/std": 0.3458506464958191, "rewards/total_composite/mean": 0.799231767654419, "rewards/total_composite/std": 0.3458506464958191, "sampling/importance_sampling_ratio/max": 1.8707276582717896, "sampling/importance_sampling_ratio/mean": 0.9987403750419617, "sampling/importance_sampling_ratio/min": 0.34747663140296936, "sampling/sampling_logp_difference/max": 1.0570578575134277, "sampling/sampling_logp_difference/mean": 0.07469379901885986, "step": 419 }, { "clip_ratio/high_max": 0.014262907905504107, "clip_ratio/high_mean": 0.014262907905504107, "clip_ratio/low_mean": 0.006358225247822702, "clip_ratio/low_min": 0.006358225247822702, "clip_ratio/region_mean": 0.02062113315332681, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 72.75, "completions/mean_terminated_length": 72.75, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.13834022916853428, "epoch": 0.016218721037998145, "frac_reward_zero_std": 0.0, "grad_norm": 4.019112586975098, "learning_rate": 8.73030303030303e-06, "loss": 0.0655, "num_tokens": 906465.0, "reward": 0.9897938966751099, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9897938966751099, "reward_meter_std": 0.00681549496948719, "reward_std": 0.006815491709858179, "reward_total_composite_mean": 0.9897938966751099, "reward_total_composite_std": 0.00681549496948719, "reward_total_mean": 0.9897938966751099, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9897938966751099, "rewards/meter/std": 0.00681549496948719, "rewards/total_composite/mean": 0.9897938966751099, "rewards/total_composite/std": 0.00681549496948719, "sampling/importance_sampling_ratio/max": 1.771165132522583, "sampling/importance_sampling_ratio/mean": 1.0009511709213257, "sampling/importance_sampling_ratio/min": 0.07870874553918839, "sampling/sampling_logp_difference/max": 2.5420010089874268, "sampling/sampling_logp_difference/mean": 0.030092770233750343, "step": 420 }, { "clip_ratio/high_max": 0.016083280788734555, "clip_ratio/high_mean": 0.016083280788734555, "clip_ratio/low_mean": 0.005128088872879744, "clip_ratio/low_min": 0.005128088872879744, "clip_ratio/region_mean": 0.0212113696616143, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 157.0, "completions/mean_length": 189.125, "completions/mean_terminated_length": 143.0, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "entropy": 0.14536871574819088, "epoch": 0.016257337040469572, "frac_reward_zero_std": 0.0, "grad_norm": 2.0794014930725098, "learning_rate": 8.727272727272728e-06, "loss": -0.119, "num_tokens": 908794.0, "reward": 0.6646036505699158, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_meter_mean": 0.6905167698860168, "reward_meter_std": 0.3626475930213928, "reward_std": 0.4017622768878937, "reward_total_composite_mean": 0.6646036505699158, "reward_total_composite_std": 0.4017622768878937, "reward_total_mean": 0.6646036505699158, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/meter/mean": 0.6905167698860168, "rewards/meter/std": 0.3626475930213928, "rewards/total_composite/mean": 0.6646036505699158, "rewards/total_composite/std": 0.4017622768878937, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0038185119628906, "sampling/importance_sampling_ratio/min": 0.39857566356658936, "sampling/sampling_logp_difference/max": 0.9549951553344727, "sampling/sampling_logp_difference/mean": 0.02924906462430954, "step": 421 }, { "clip_ratio/high_max": 0.04184396180789918, "clip_ratio/high_mean": 0.04184396180789918, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/region_mean": 0.046229926752857864, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 121.375, "completions/mean_terminated_length": 121.375, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.49255986511707306, "epoch": 0.016295953042940996, "frac_reward_zero_std": 0.0, "grad_norm": 4.160637855529785, "learning_rate": 8.724242424242425e-06, "loss": -0.026, "num_tokens": 911117.0, "reward": 0.9373536109924316, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9373536109924316, "reward_meter_std": 0.09886158257722855, "reward_std": 0.09886158257722855, "reward_total_composite_mean": 0.9373536109924316, "reward_total_composite_std": 0.09886158257722855, "reward_total_mean": 0.9373536109924316, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9373536109924316, "rewards/meter/std": 0.09886158257722855, "rewards/total_composite/mean": 0.9373536109924316, "rewards/total_composite/std": 0.09886158257722855, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.011407732963562, "sampling/importance_sampling_ratio/min": 0.25942739844322205, "sampling/sampling_logp_difference/max": 1.349278450012207, "sampling/sampling_logp_difference/mean": 0.05337180942296982, "step": 422 }, { "clip_ratio/high_max": 0.01431451621465385, "clip_ratio/high_mean": 0.01431451621465385, "clip_ratio/low_mean": 0.018862260156311095, "clip_ratio/low_min": 0.018862260156311095, "clip_ratio/region_mean": 0.033176776370964944, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.16816303879022598, "epoch": 0.01633456904541242, "frac_reward_zero_std": 0.0, "grad_norm": 13.163901329040527, "learning_rate": 8.72121212121212e-06, "loss": 0.0591, "num_tokens": 912910.0, "reward": 0.9770172834396362, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9770172834396362, "reward_meter_std": 0.024198686704039574, "reward_std": 0.024198684841394424, "reward_total_composite_mean": 0.9770172834396362, "reward_total_composite_std": 0.024198686704039574, "reward_total_mean": 0.9770172834396362, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9770172834396362, "rewards/meter/std": 0.024198686704039574, "rewards/total_composite/mean": 0.9770172834396362, "rewards/total_composite/std": 0.024198686704039574, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003563642501831, "sampling/importance_sampling_ratio/min": 0.03152162954211235, "sampling/sampling_logp_difference/max": 3.4570813179016113, "sampling/sampling_logp_difference/mean": 0.04488004371523857, "step": 423 }, { "clip_ratio/high_max": 0.04446423542685807, "clip_ratio/high_mean": 0.04446423542685807, "clip_ratio/low_mean": 0.013393073342740536, "clip_ratio/low_min": 0.013393073342740536, "clip_ratio/region_mean": 0.0578573087695986, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 78.0, "completions/mean_terminated_length": 78.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.281008530408144, "epoch": 0.016373185047883845, "frac_reward_zero_std": 0.0, "grad_norm": 6.008713722229004, "learning_rate": 8.71818181818182e-06, "loss": 0.0378, "num_tokens": 914750.0, "reward": 0.7533670663833618, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7533670663833618, "reward_meter_std": 0.23079943656921387, "reward_std": 0.23079942166805267, "reward_total_composite_mean": 0.7533670663833618, "reward_total_composite_std": 0.23079943656921387, "reward_total_mean": 0.7533670663833618, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7533670663833618, "rewards/meter/std": 0.23079943656921387, "rewards/total_composite/mean": 0.7533670663833618, "rewards/total_composite/std": 0.23079943656921387, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9994415044784546, "sampling/importance_sampling_ratio/min": 0.012960345484316349, "sampling/sampling_logp_difference/max": 4.345860958099365, "sampling/sampling_logp_difference/mean": 0.0553349107503891, "step": 424 }, { "clip_ratio/high_max": 0.025681341998279095, "clip_ratio/high_mean": 0.025681341998279095, "clip_ratio/low_mean": 0.05649581435136497, "clip_ratio/low_min": 0.05649581435136497, "clip_ratio/region_mean": 0.08217715634964406, "completions/clipped_ratio": 0.0, "completions/max_length": 53.0, "completions/max_terminated_length": 53.0, "completions/mean_length": 45.625, "completions/mean_terminated_length": 45.625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.5226194709539413, "epoch": 0.01641180105035527, "frac_reward_zero_std": 0.0, "grad_norm": 8.99482536315918, "learning_rate": 8.715151515151515e-06, "loss": 0.0023, "num_tokens": 916259.0, "reward": 0.18696027994155884, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.18696027994155884, "reward_meter_std": 0.2777934968471527, "reward_std": 0.2777934968471527, "reward_total_composite_mean": 0.18696027994155884, "reward_total_composite_std": 0.2777934968471527, "reward_total_mean": 0.18696027994155884, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.18696027994155884, "rewards/meter/std": 0.2777934968471527, "rewards/total_composite/mean": 0.18696027994155884, "rewards/total_composite/std": 0.2777934968471527, "sampling/importance_sampling_ratio/max": 1.9791115522384644, "sampling/importance_sampling_ratio/mean": 1.0027213096618652, "sampling/importance_sampling_ratio/min": 0.2798304259777069, "sampling/sampling_logp_difference/max": 1.273571491241455, "sampling/sampling_logp_difference/mean": 0.08858486264944077, "step": 425 }, { "clip_ratio/high_max": 0.04074844322167337, "clip_ratio/high_mean": 0.04074844322167337, "clip_ratio/low_mean": 0.018849206971935928, "clip_ratio/low_min": 0.018849206971935928, "clip_ratio/region_mean": 0.0595976501936093, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.481033593416214, "epoch": 0.016450417052826693, "frac_reward_zero_std": 0.0, "grad_norm": 5.837379455566406, "learning_rate": 8.712121212121212e-06, "loss": 0.0217, "num_tokens": 918078.0, "reward": 0.7051376700401306, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7622324228286743, "reward_meter_std": 0.3091115951538086, "reward_std": 0.31919535994529724, "reward_total_composite_mean": 0.7051376700401306, "reward_total_composite_std": 0.31919533014297485, "reward_total_mean": 0.7051376700401306, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7622324228286743, "rewards/meter/std": 0.3091115951538086, "rewards/total_composite/mean": 0.7051376700401306, "rewards/total_composite/std": 0.31919533014297485, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010563611984253, "sampling/importance_sampling_ratio/min": 0.030512982979416847, "sampling/sampling_logp_difference/max": 3.489603042602539, "sampling/sampling_logp_difference/mean": 0.07148178666830063, "step": 426 }, { "clip_ratio/high_max": 0.023971143178641796, "clip_ratio/high_mean": 0.023971143178641796, "clip_ratio/low_mean": 0.009528988506644964, "clip_ratio/low_min": 0.009528988506644964, "clip_ratio/region_mean": 0.03350013168528676, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 150.25, "completions/mean_terminated_length": 150.25, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.14678781293332577, "epoch": 0.016489033055298117, "frac_reward_zero_std": 0.0, "grad_norm": 4.158021450042725, "learning_rate": 8.70909090909091e-06, "loss": -0.0653, "num_tokens": 920736.0, "reward": 0.6173264980316162, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.13363061845302582, "reward_meter_mean": 0.6729066967964172, "reward_meter_std": 0.3364524841308594, "reward_std": 0.3513941466808319, "reward_total_composite_mean": 0.6173264980316162, "reward_total_composite_std": 0.3513941466808319, "reward_total_mean": 0.6173264980316162, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.13363061845302582, "rewards/meter/mean": 0.6729066967964172, "rewards/meter/std": 0.3364524841308594, "rewards/total_composite/mean": 0.6173264980316162, "rewards/total_composite/std": 0.3513941466808319, "sampling/importance_sampling_ratio/max": 1.959710717201233, "sampling/importance_sampling_ratio/mean": 0.9964794516563416, "sampling/importance_sampling_ratio/min": 0.05687982589006424, "sampling/sampling_logp_difference/max": 2.866814613342285, "sampling/sampling_logp_difference/mean": 0.03079906478524208, "step": 427 }, { "clip_ratio/high_max": 0.010739810299128294, "clip_ratio/high_mean": 0.010739810299128294, "clip_ratio/low_mean": 0.009472908801399171, "clip_ratio/low_min": 0.009472908801399171, "clip_ratio/region_mean": 0.020212719100527465, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 129.0, "completions/mean_terminated_length": 129.0, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.13649542536586523, "epoch": 0.01652764905776954, "frac_reward_zero_std": 0.0, "grad_norm": 2.6976351737976074, "learning_rate": 8.706060606060607e-06, "loss": 0.0675, "num_tokens": 923000.0, "reward": 0.11112642288208008, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.11112642288208008, "reward_meter_std": 0.1817607432603836, "reward_std": 0.1817607432603836, "reward_total_composite_mean": 0.11112642288208008, "reward_total_composite_std": 0.1817607432603836, "reward_total_mean": 0.11112642288208008, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.11112642288208008, "rewards/meter/std": 0.1817607432603836, "rewards/total_composite/mean": 0.11112642288208008, "rewards/total_composite/std": 0.1817607432603836, "sampling/importance_sampling_ratio/max": 1.9328269958496094, "sampling/importance_sampling_ratio/mean": 1.0000661611557007, "sampling/importance_sampling_ratio/min": 0.07731583714485168, "sampling/sampling_logp_difference/max": 2.559856414794922, "sampling/sampling_logp_difference/mean": 0.030291767790913582, "step": 428 }, { "clip_ratio/high_max": 0.02691519889049232, "clip_ratio/high_mean": 0.02691519889049232, "clip_ratio/low_mean": 0.012388192000798881, "clip_ratio/low_min": 0.012388192000798881, "clip_ratio/region_mean": 0.0393033908912912, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 127.125, "completions/mean_terminated_length": 127.125, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.26551887206733227, "epoch": 0.016566265060240965, "frac_reward_zero_std": 0.0, "grad_norm": 3.4749865531921387, "learning_rate": 8.703030303030304e-06, "loss": -0.0293, "num_tokens": 925409.0, "reward": 0.8244721293449402, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_meter_mean": 0.9445109963417053, "reward_meter_std": 0.06675712764263153, "reward_std": 0.16515542566776276, "reward_total_composite_mean": 0.8244721293449402, "reward_total_composite_std": 0.16515542566776276, "reward_total_mean": 0.8244721293449402, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/meter/mean": 0.9445109963417053, "rewards/meter/std": 0.06675712764263153, "rewards/total_composite/mean": 0.8244721293449402, "rewards/total_composite/std": 0.16515542566776276, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9982576966285706, "sampling/importance_sampling_ratio/min": 0.019715862348675728, "sampling/sampling_logp_difference/max": 3.9263317584991455, "sampling/sampling_logp_difference/mean": 0.05006590485572815, "step": 429 }, { "clip_ratio/high_max": 0.03724813973531127, "clip_ratio/high_mean": 0.03724813973531127, "clip_ratio/low_mean": 0.03095380635932088, "clip_ratio/low_min": 0.03095380635932088, "clip_ratio/region_mean": 0.06820194609463215, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.5216472372412682, "epoch": 0.01660488106271239, "frac_reward_zero_std": 0.0, "grad_norm": 5.7448906898498535, "learning_rate": 8.700000000000001e-06, "loss": 0.0512, "num_tokens": 927497.0, "reward": 0.574688196182251, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.574688196182251, "reward_meter_std": 0.4255768954753876, "reward_std": 0.42557692527770996, "reward_total_composite_mean": 0.574688196182251, "reward_total_composite_std": 0.4255768954753876, "reward_total_mean": 0.574688196182251, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.574688196182251, "rewards/meter/std": 0.4255768954753876, "rewards/total_composite/mean": 0.574688196182251, "rewards/total_composite/std": 0.4255768954753876, "sampling/importance_sampling_ratio/max": 1.9886913299560547, "sampling/importance_sampling_ratio/mean": 1.0072433948516846, "sampling/importance_sampling_ratio/min": 0.3208860456943512, "sampling/sampling_logp_difference/max": 1.1366691589355469, "sampling/sampling_logp_difference/mean": 0.06919876486063004, "step": 430 }, { "clip_ratio/high_max": 0.060504904482513666, "clip_ratio/high_mean": 0.060504904482513666, "clip_ratio/low_mean": 0.022997836116701365, "clip_ratio/low_min": 0.022997836116701365, "clip_ratio/region_mean": 0.08350274059921503, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 34.125, "completions/mean_terminated_length": 34.125, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.9744241572916508, "epoch": 0.016643497065183813, "frac_reward_zero_std": 0.0, "grad_norm": 14.200037956237793, "learning_rate": 8.696969696969699e-06, "loss": -0.0277, "num_tokens": 929098.0, "reward": 0.811887264251709, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.811887264251709, "reward_meter_std": 0.2896690368652344, "reward_std": 0.2896690368652344, "reward_total_composite_mean": 0.811887264251709, "reward_total_composite_std": 0.2896690368652344, "reward_total_mean": 0.811887264251709, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.811887264251709, "rewards/meter/std": 0.2896690368652344, "rewards/total_composite/mean": 0.811887264251709, "rewards/total_composite/std": 0.2896690368652344, "sampling/importance_sampling_ratio/max": 1.7877914905548096, "sampling/importance_sampling_ratio/mean": 1.0314644575119019, "sampling/importance_sampling_ratio/min": 0.440626323223114, "sampling/sampling_logp_difference/max": 0.8195581436157227, "sampling/sampling_logp_difference/mean": 0.09828025102615356, "step": 431 }, { "clip_ratio/high_max": 0.0062806373462080956, "clip_ratio/high_mean": 0.0062806373462080956, "clip_ratio/low_mean": 0.01565172686241567, "clip_ratio/low_min": 0.01565172686241567, "clip_ratio/region_mean": 0.021932364208623767, "completions/clipped_ratio": 0.0, "completions/max_length": 113.0, "completions/max_terminated_length": 113.0, "completions/mean_length": 103.0, "completions/mean_terminated_length": 103.0, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.12999565247446299, "epoch": 0.016682113067655237, "frac_reward_zero_std": 0.0, "grad_norm": 6.835657596588135, "learning_rate": 8.693939393939394e-06, "loss": 0.038, "num_tokens": 931202.0, "reward": 0.45791012048721313, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.45791012048721313, "reward_meter_std": 0.43955934047698975, "reward_std": 0.43955934047698975, "reward_total_composite_mean": 0.45791012048721313, "reward_total_composite_std": 0.43955934047698975, "reward_total_mean": 0.45791012048721313, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.45791012048721313, "rewards/meter/std": 0.43955934047698975, "rewards/total_composite/mean": 0.45791012048721313, "rewards/total_composite/std": 0.43955934047698975, "sampling/importance_sampling_ratio/max": 1.681004524230957, "sampling/importance_sampling_ratio/mean": 0.9998974800109863, "sampling/importance_sampling_ratio/min": 0.3137332797050476, "sampling/sampling_logp_difference/max": 1.1592121124267578, "sampling/sampling_logp_difference/mean": 0.025745589286088943, "step": 432 }, { "clip_ratio/high_max": 0.03764841705560684, "clip_ratio/high_mean": 0.03764841705560684, "clip_ratio/low_mean": 0.0251602572388947, "clip_ratio/low_min": 0.0251602572388947, "clip_ratio/region_mean": 0.06280867429450154, "completions/clipped_ratio": 0.0, "completions/max_length": 110.0, "completions/max_terminated_length": 110.0, "completions/mean_length": 94.875, "completions/mean_terminated_length": 94.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.45340409502387047, "epoch": 0.01672072907012666, "frac_reward_zero_std": 0.0, "grad_norm": 9.083555221557617, "learning_rate": 8.690909090909091e-06, "loss": 0.0363, "num_tokens": 933225.0, "reward": 0.7613606452941895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.8027656674385071, "reward_meter_std": 0.3288112282752991, "reward_std": 0.3221178352832794, "reward_total_composite_mean": 0.7613606452941895, "reward_total_composite_std": 0.3221178352832794, "reward_total_mean": 0.7613606452941895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.8027656674385071, "rewards/meter/std": 0.3288112282752991, "rewards/total_composite/mean": 0.7613606452941895, "rewards/total_composite/std": 0.3221178352832794, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0011898279190063, "sampling/importance_sampling_ratio/min": 0.2318556159734726, "sampling/sampling_logp_difference/max": 1.461640477180481, "sampling/sampling_logp_difference/mean": 0.08027959614992142, "step": 433 }, { "clip_ratio/high_max": 0.013409517356194556, "clip_ratio/high_mean": 0.013409517356194556, "clip_ratio/low_mean": 0.0010869564721360803, "clip_ratio/low_min": 0.0010869564721360803, "clip_ratio/region_mean": 0.014496473828330636, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 114.5, "completions/mean_terminated_length": 114.5, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.11784437298774719, "epoch": 0.016759345072598086, "frac_reward_zero_std": 0.0, "grad_norm": 3.4593441486358643, "learning_rate": 8.687878787878789e-06, "loss": 0.007, "num_tokens": 935461.0, "reward": 0.8310332298278809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8310332298278809, "reward_meter_std": 0.32082754373550415, "reward_std": 0.32082754373550415, "reward_total_composite_mean": 0.8310332298278809, "reward_total_composite_std": 0.32082754373550415, "reward_total_mean": 0.8310332298278809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8310332298278809, "rewards/meter/std": 0.32082754373550415, "rewards/total_composite/mean": 0.8310332298278809, "rewards/total_composite/std": 0.32082754373550415, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0050222873687744, "sampling/importance_sampling_ratio/min": 0.0013107022969052196, "sampling/sampling_logp_difference/max": 6.637192249298096, "sampling/sampling_logp_difference/mean": 0.032745786011219025, "step": 434 }, { "clip_ratio/high_max": 0.04601668380200863, "clip_ratio/high_mean": 0.04601668380200863, "clip_ratio/low_mean": 0.014399510342627764, "clip_ratio/low_min": 0.014399510342627764, "clip_ratio/region_mean": 0.06041619414463639, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 101.125, "completions/mean_terminated_length": 101.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.5314316283911467, "epoch": 0.01679796107506951, "frac_reward_zero_std": 0.0, "grad_norm": 5.847456455230713, "learning_rate": 8.684848484848486e-06, "loss": -0.0446, "num_tokens": 937614.0, "reward": 0.8126308917999268, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.8531937599182129, "reward_meter_std": 0.27812930941581726, "reward_std": 0.2817155420780182, "reward_total_composite_mean": 0.8126308917999268, "reward_total_composite_std": 0.2817155122756958, "reward_total_mean": 0.8126308917999268, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.8531937599182129, "rewards/meter/std": 0.27812930941581726, "rewards/total_composite/mean": 0.8126308917999268, "rewards/total_composite/std": 0.2817155122756958, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002807378768921, "sampling/importance_sampling_ratio/min": 0.12069769948720932, "sampling/sampling_logp_difference/max": 2.1144661903381348, "sampling/sampling_logp_difference/mean": 0.07007157802581787, "step": 435 }, { "clip_ratio/high_max": 0.0015625000232830644, "clip_ratio/high_mean": 0.0015625000232830644, "clip_ratio/low_mean": 0.026244841050356627, "clip_ratio/low_min": 0.026244841050356627, "clip_ratio/region_mean": 0.02780734107363969, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 77.25, "completions/mean_terminated_length": 77.25, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.1496438980102539, "epoch": 0.016836577077540934, "frac_reward_zero_std": 0.0, "grad_norm": 5.732955455780029, "learning_rate": 8.681818181818182e-06, "loss": -0.0109, "num_tokens": 939392.0, "reward": 0.3143185079097748, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3143185079097748, "reward_meter_std": 0.41200879216194153, "reward_std": 0.41200879216194153, "reward_total_composite_mean": 0.3143185079097748, "reward_total_composite_std": 0.41200879216194153, "reward_total_mean": 0.3143185079097748, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3143185079097748, "rewards/meter/std": 0.41200879216194153, "rewards/total_composite/mean": 0.3143185079097748, "rewards/total_composite/std": 0.41200879216194153, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0097142457962036, "sampling/importance_sampling_ratio/min": 0.01733795367181301, "sampling/sampling_logp_difference/max": 4.05485725402832, "sampling/sampling_logp_difference/mean": 0.04445616900920868, "step": 436 }, { "clip_ratio/high_max": 0.014880952425301075, "clip_ratio/high_mean": 0.014880952425301075, "clip_ratio/low_mean": 0.01840795623138547, "clip_ratio/low_min": 0.01840795623138547, "clip_ratio/region_mean": 0.033288908656686544, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 82.75, "completions/mean_terminated_length": 82.75, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.3760291952639818, "epoch": 0.016875193080012358, "frac_reward_zero_std": 0.0, "grad_norm": 4.509356498718262, "learning_rate": 8.67878787878788e-06, "loss": -0.0275, "num_tokens": 941350.0, "reward": 0.2339237928390503, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_meter_mean": 0.2568199038505554, "reward_meter_std": 0.359527051448822, "reward_std": 0.3625546097755432, "reward_total_composite_mean": 0.2339237928390503, "reward_total_composite_std": 0.3625546097755432, "reward_total_mean": 0.2339237928390503, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/meter/mean": 0.2568199038505554, "rewards/meter/std": 0.359527051448822, "rewards/total_composite/mean": 0.2339237928390503, "rewards/total_composite/std": 0.3625546097755432, "sampling/importance_sampling_ratio/max": 1.8314160108566284, "sampling/importance_sampling_ratio/mean": 1.0075230598449707, "sampling/importance_sampling_ratio/min": 0.25561147928237915, "sampling/sampling_logp_difference/max": 1.3640966415405273, "sampling/sampling_logp_difference/mean": 0.058143459260463715, "step": 437 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.016913809082483782, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 8.675757575757576e-06, "loss": 0.0, "num_tokens": 943046.0, "reward": 0.7461053729057312, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948071241378784, "reward_meter_std": 0.00352023309096694, "reward_std": 0.0026401823852211237, "reward_total_composite_mean": 0.7461053729057312, "reward_total_composite_std": 0.0026401823852211237, "reward_total_mean": 0.7461053729057312, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948071241378784, "rewards/meter/std": 0.00352023309096694, "rewards/total_composite/mean": 0.7461053729057312, "rewards/total_composite/std": 0.0026401823852211237, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 438 }, { "clip_ratio/high_max": 0.023564013885334134, "clip_ratio/high_mean": 0.023564013885334134, "clip_ratio/low_mean": 0.006527067394927144, "clip_ratio/low_min": 0.006527067394927144, "clip_ratio/region_mean": 0.030091081280261278, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 94.25, "completions/mean_terminated_length": 94.25, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.20515940617769957, "epoch": 0.016952425084955206, "frac_reward_zero_std": 0.0, "grad_norm": 3.7757630348205566, "learning_rate": 8.672727272727273e-06, "loss": 0.0153, "num_tokens": 945072.0, "reward": 0.7195857763290405, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8440791368484497, "reward_meter_std": 0.29224100708961487, "reward_std": 0.4076501131057739, "reward_total_composite_mean": 0.7195857763290405, "reward_total_composite_std": 0.4076501429080963, "reward_total_mean": 0.7195857763290405, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8440791368484497, "rewards/meter/std": 0.29224100708961487, "rewards/total_composite/mean": 0.7195857763290405, "rewards/total_composite/std": 0.4076501429080963, "sampling/importance_sampling_ratio/max": 1.6128379106521606, "sampling/importance_sampling_ratio/mean": 1.0024596452713013, "sampling/importance_sampling_ratio/min": 0.1524176001548767, "sampling/sampling_logp_difference/max": 1.8811311721801758, "sampling/sampling_logp_difference/mean": 0.031045421957969666, "step": 439 }, { "clip_ratio/high_max": 0.008545128046534956, "clip_ratio/high_mean": 0.008545128046534956, "clip_ratio/low_mean": 0.013955606264062226, "clip_ratio/low_min": 0.013955606264062226, "clip_ratio/region_mean": 0.02250073431059718, "completions/clipped_ratio": 0.0, "completions/max_length": 223.0, "completions/max_terminated_length": 223.0, "completions/mean_length": 199.25, "completions/mean_terminated_length": 199.25, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 0.1384673686698079, "epoch": 0.01699104108742663, "frac_reward_zero_std": 0.0, "grad_norm": 4.999176979064941, "learning_rate": 8.66969696969697e-06, "loss": -0.0297, "num_tokens": 948306.0, "reward": 0.800205647945404, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.1035098284482956, "reward_meter_mean": 0.906821608543396, "reward_meter_std": 0.20782814919948578, "reward_std": 0.22486314177513123, "reward_total_composite_mean": 0.800205647945404, "reward_total_composite_std": 0.22486315667629242, "reward_total_mean": 0.800205647945404, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.1035098284482956, "rewards/meter/mean": 0.906821608543396, "rewards/meter/std": 0.20782814919948578, "rewards/total_composite/mean": 0.800205647945404, "rewards/total_composite/std": 0.22486315667629242, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0032387971878052, "sampling/importance_sampling_ratio/min": 0.13524653017520905, "sampling/sampling_logp_difference/max": 2.0006561279296875, "sampling/sampling_logp_difference/mean": 0.024938292801380157, "step": 440 }, { "clip_ratio/high_max": 0.06254499591886997, "clip_ratio/high_mean": 0.06254499591886997, "clip_ratio/low_mean": 0.04578754771500826, "clip_ratio/low_min": 0.04578754771500826, "clip_ratio/region_mean": 0.10833254363387823, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 33.625, "completions/mean_terminated_length": 33.625, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.6665975973010063, "epoch": 0.017029657089898054, "frac_reward_zero_std": 0.0, "grad_norm": 12.705753326416016, "learning_rate": 8.666666666666668e-06, "loss": 0.043, "num_tokens": 949895.0, "reward": 0.7382475137710571, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7382475137710571, "reward_meter_std": 0.344305157661438, "reward_std": 0.344305157661438, "reward_total_composite_mean": 0.7382475137710571, "reward_total_composite_std": 0.344305157661438, "reward_total_mean": 0.7382475137710571, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7382475137710571, "rewards/meter/std": 0.344305157661438, "rewards/total_composite/mean": 0.7382475137710571, "rewards/total_composite/std": 0.344305157661438, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0123142004013062, "sampling/importance_sampling_ratio/min": 0.2521496117115021, "sampling/sampling_logp_difference/max": 1.377732753753662, "sampling/sampling_logp_difference/mean": 0.11312177032232285, "step": 441 }, { "clip_ratio/high_max": 0.04311846289783716, "clip_ratio/high_mean": 0.04311846289783716, "clip_ratio/low_mean": 0.018479025457054377, "clip_ratio/low_min": 0.018479025457054377, "clip_ratio/region_mean": 0.06159748835489154, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 67.375, "completions/mean_terminated_length": 67.375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.44354628771543503, "epoch": 0.01706827309236948, "frac_reward_zero_std": 0.0, "grad_norm": 6.052305698394775, "learning_rate": 8.663636363636363e-06, "loss": 0.0748, "num_tokens": 951634.0, "reward": 0.6851848363876343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6851848363876343, "reward_meter_std": 0.40763622522354126, "reward_std": 0.40763622522354126, "reward_total_composite_mean": 0.6851848363876343, "reward_total_composite_std": 0.40763622522354126, "reward_total_mean": 0.6851848363876343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6851848363876343, "rewards/meter/std": 0.40763622522354126, "rewards/total_composite/mean": 0.6851848363876343, "rewards/total_composite/std": 0.40763622522354126, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01041579246521, "sampling/importance_sampling_ratio/min": 0.14436128735542297, "sampling/sampling_logp_difference/max": 1.9354361295700073, "sampling/sampling_logp_difference/mean": 0.06014850363135338, "step": 442 }, { "clip_ratio/high_max": 0.04367695190012455, "clip_ratio/high_mean": 0.04367695190012455, "clip_ratio/low_mean": 0.014395925216376781, "clip_ratio/low_min": 0.014395925216376781, "clip_ratio/region_mean": 0.05807287711650133, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 68.625, "completions/mean_terminated_length": 68.625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.5356311798095703, "epoch": 0.017106889094840903, "frac_reward_zero_std": 0.0, "grad_norm": 9.460426330566406, "learning_rate": 8.660606060606062e-06, "loss": 0.0122, "num_tokens": 953439.0, "reward": 0.861977219581604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.861977219581604, "reward_meter_std": 0.24617436528205872, "reward_std": 0.24617433547973633, "reward_total_composite_mean": 0.861977219581604, "reward_total_composite_std": 0.24617436528205872, "reward_total_mean": 0.861977219581604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.861977219581604, "rewards/meter/std": 0.24617436528205872, "rewards/total_composite/mean": 0.861977219581604, "rewards/total_composite/std": 0.24617436528205872, "sampling/importance_sampling_ratio/max": 1.7051851749420166, "sampling/importance_sampling_ratio/mean": 1.0046181678771973, "sampling/importance_sampling_ratio/min": 0.1390223354101181, "sampling/sampling_logp_difference/max": 1.9731206893920898, "sampling/sampling_logp_difference/mean": 0.06774439662694931, "step": 443 }, { "clip_ratio/high_max": 0.011721267364919186, "clip_ratio/high_mean": 0.011721267364919186, "clip_ratio/low_mean": 0.05329327145591378, "clip_ratio/low_min": 0.05329327145591378, "clip_ratio/region_mean": 0.06501453882083297, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 106.625, "completions/mean_terminated_length": 106.625, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.31410662829875946, "epoch": 0.017145505097312327, "frac_reward_zero_std": 0.0, "grad_norm": 6.638420104980469, "learning_rate": 8.657575757575758e-06, "loss": 0.0611, "num_tokens": 955748.0, "reward": 0.24504423141479492, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.24504423141479492, "reward_meter_std": 0.346823513507843, "reward_std": 0.346823513507843, "reward_total_composite_mean": 0.24504423141479492, "reward_total_composite_std": 0.346823513507843, "reward_total_mean": 0.24504423141479492, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.24504423141479492, "rewards/meter/std": 0.346823513507843, "rewards/total_composite/mean": 0.24504423141479492, "rewards/total_composite/std": 0.346823513507843, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0113799571990967, "sampling/importance_sampling_ratio/min": 0.2446102499961853, "sampling/sampling_logp_difference/max": 1.4080891609191895, "sampling/sampling_logp_difference/mean": 0.05710998922586441, "step": 444 }, { "clip_ratio/high_max": 0.05313561297953129, "clip_ratio/high_mean": 0.05313561297953129, "clip_ratio/low_mean": 0.021353119518607855, "clip_ratio/low_min": 0.021353119518607855, "clip_ratio/region_mean": 0.07448873249813914, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.7675438597798347, "epoch": 0.01718412109978375, "frac_reward_zero_std": 0.0, "grad_norm": 9.578984260559082, "learning_rate": 8.654545454545455e-06, "loss": 0.0332, "num_tokens": 957485.0, "reward": 0.7745425701141357, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7745425701141357, "reward_meter_std": 0.3829474449157715, "reward_std": 0.38294747471809387, "reward_total_composite_mean": 0.7745425701141357, "reward_total_composite_std": 0.3829474449157715, "reward_total_mean": 0.7745425701141357, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7745425701141357, "rewards/meter/std": 0.3829474449157715, "rewards/total_composite/mean": 0.7745425701141357, "rewards/total_composite/std": 0.3829474449157715, "sampling/importance_sampling_ratio/max": 1.9029569625854492, "sampling/importance_sampling_ratio/mean": 1.0089476108551025, "sampling/importance_sampling_ratio/min": 0.31309664249420166, "sampling/sampling_logp_difference/max": 1.1612434387207031, "sampling/sampling_logp_difference/mean": 0.08342333137989044, "step": 445 }, { "clip_ratio/high_max": 0.05887592723593116, "clip_ratio/high_mean": 0.05887592723593116, "clip_ratio/low_mean": 0.02302631549537182, "clip_ratio/low_min": 0.02302631549537182, "clip_ratio/region_mean": 0.08190224273130298, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 38.375, "completions/mean_terminated_length": 38.375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.6561026684939861, "epoch": 0.017222737102255175, "frac_reward_zero_std": 0.0, "grad_norm": 16.04683494567871, "learning_rate": 8.651515151515152e-06, "loss": 0.0135, "num_tokens": 959040.0, "reward": 0.8562139272689819, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8562139272689819, "reward_meter_std": 0.34138572216033936, "reward_std": 0.34138569235801697, "reward_total_composite_mean": 0.8562139272689819, "reward_total_composite_std": 0.34138572216033936, "reward_total_mean": 0.8562139272689819, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8562139272689819, "rewards/meter/std": 0.34138572216033936, "rewards/total_composite/mean": 0.8562139272689819, "rewards/total_composite/std": 0.34138572216033936, "sampling/importance_sampling_ratio/max": 1.844504952430725, "sampling/importance_sampling_ratio/mean": 1.0087430477142334, "sampling/importance_sampling_ratio/min": 0.2737632095813751, "sampling/sampling_logp_difference/max": 1.2954916954040527, "sampling/sampling_logp_difference/mean": 0.09310141205787659, "step": 446 }, { "clip_ratio/high_max": 0.011402326985262334, "clip_ratio/high_mean": 0.011402326985262334, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.011402326985262334, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 197.0, "completions/mean_length": 230.375, "completions/mean_terminated_length": 190.1428680419922, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.12354243081063032, "epoch": 0.0172613531047266, "frac_reward_zero_std": 0.0, "grad_norm": 0.7952864766120911, "learning_rate": 8.64848484848485e-06, "loss": -0.2555, "num_tokens": 961699.0, "reward": 0.8865582346916199, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_meter_mean": 0.9410502910614014, "reward_meter_std": 0.1454276442527771, "reward_std": 0.29953086376190186, "reward_total_composite_mean": 0.8865582346916199, "reward_total_composite_std": 0.29953092336654663, "reward_total_mean": 0.8865582346916199, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/meter/mean": 0.9410502910614014, "rewards/meter/std": 0.1454276442527771, "rewards/total_composite/mean": 0.8865582346916199, "rewards/total_composite/std": 0.29953092336654663, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0042375326156616, "sampling/importance_sampling_ratio/min": 0.3248317837715149, "sampling/sampling_logp_difference/max": 1.1244478225708008, "sampling/sampling_logp_difference/mean": 0.01994623802602291, "step": 447 }, { "clip_ratio/high_max": 0.01393397233914584, "clip_ratio/high_mean": 0.01393397233914584, "clip_ratio/low_mean": 0.002027027076110244, "clip_ratio/low_min": 0.002027027076110244, "clip_ratio/region_mean": 0.015960999415256083, "completions/clipped_ratio": 0.0, "completions/max_length": 185.0, "completions/max_terminated_length": 185.0, "completions/mean_length": 158.75, "completions/mean_terminated_length": 158.75, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.10146280331537127, "epoch": 0.017299969107198023, "frac_reward_zero_std": 0.0, "grad_norm": 2.0035412311553955, "learning_rate": 8.645454545454545e-06, "loss": 0.0638, "num_tokens": 964297.0, "reward": 0.859869122505188, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.859869122505188, "reward_meter_std": 0.3417814075946808, "reward_std": 0.3417814075946808, "reward_total_composite_mean": 0.859869122505188, "reward_total_composite_std": 0.3417814075946808, "reward_total_mean": 0.859869122505188, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.859869122505188, "rewards/meter/std": 0.3417814075946808, "rewards/total_composite/mean": 0.859869122505188, "rewards/total_composite/std": 0.3417814075946808, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9988955855369568, "sampling/importance_sampling_ratio/min": 0.005555473268032074, "sampling/sampling_logp_difference/max": 5.192971706390381, "sampling/sampling_logp_difference/mean": 0.03218178451061249, "step": 448 }, { "clip_ratio/high_max": 0.029356154147535563, "clip_ratio/high_mean": 0.029356154147535563, "clip_ratio/low_mean": 0.007736280560493469, "clip_ratio/low_min": 0.007736280560493469, "clip_ratio/region_mean": 0.03709243470802903, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 82.75, "completions/mean_terminated_length": 82.75, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.2209324724972248, "epoch": 0.017338585109669447, "frac_reward_zero_std": 0.0, "grad_norm": 6.3059892654418945, "learning_rate": 8.642424242424242e-06, "loss": 0.0171, "num_tokens": 966255.0, "reward": 0.6425033807754517, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6425033807754517, "reward_meter_std": 0.41706737875938416, "reward_std": 0.41706737875938416, "reward_total_composite_mean": 0.6425033807754517, "reward_total_composite_std": 0.41706737875938416, "reward_total_mean": 0.6425033807754517, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6425033807754517, "rewards/meter/std": 0.41706737875938416, "rewards/total_composite/mean": 0.6425033807754517, "rewards/total_composite/std": 0.41706737875938416, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0040130615234375, "sampling/importance_sampling_ratio/min": 0.02065885066986084, "sampling/sampling_logp_difference/max": 3.8796114921569824, "sampling/sampling_logp_difference/mean": 0.053077779710292816, "step": 449 }, { "clip_ratio/high_max": 0.027218999108299613, "clip_ratio/high_mean": 0.027218999108299613, "clip_ratio/low_mean": 0.004411764908581972, "clip_ratio/low_min": 0.004411764908581972, "clip_ratio/region_mean": 0.031630764016881585, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 81.0, "completions/mean_terminated_length": 81.0, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.15494854096323252, "epoch": 0.01737720111214087, "frac_reward_zero_std": 0.0, "grad_norm": 6.594456195831299, "learning_rate": 8.63939393939394e-06, "loss": 0.0266, "num_tokens": 968263.0, "reward": 0.861186146736145, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.861186146736145, "reward_meter_std": 0.25013267993927, "reward_std": 0.25013265013694763, "reward_total_composite_mean": 0.861186146736145, "reward_total_composite_std": 0.25013267993927, "reward_total_mean": 0.861186146736145, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.861186146736145, "rewards/meter/std": 0.25013267993927, "rewards/total_composite/mean": 0.861186146736145, "rewards/total_composite/std": 0.25013267993927, "sampling/importance_sampling_ratio/max": 1.9482147693634033, "sampling/importance_sampling_ratio/mean": 0.9992127418518066, "sampling/importance_sampling_ratio/min": 0.1744629293680191, "sampling/sampling_logp_difference/max": 1.7460429668426514, "sampling/sampling_logp_difference/mean": 0.030779149383306503, "step": 450 }, { "epoch": 0.01737720111214087, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.10576923076923077, "eval_completions/max_length": 485.38461538461536, "eval_completions/max_terminated_length": 421.0769230769231, "eval_completions/mean_length": 252.39423076923077, "eval_completions/mean_terminated_length": 221.38736900916467, "eval_completions/min_length": 61.23076923076923, "eval_completions/min_terminated_length": 61.23076923076923, "eval_entropy": 0.11755917536524627, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 968263.0, "eval_reward": 0.6704203371818249, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.9385907145646902, "eval_reward_count_adherence_std": 0.09539258336791626, "eval_reward_meter_mean": 0.7144990059045645, "eval_reward_meter_std": 0.3781636357307434, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6704203371818249, "eval_reward_total_composite_std": 0.37348280388575333, "eval_reward_total_mean": 0.6704203371818249, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.9385907145646902, "eval_rewards/count_adherence/std": 0.09539258336791626, "eval_rewards/meter/mean": 0.7144990059045645, "eval_rewards/meter/std": 0.3781636357307434, "eval_rewards/total_composite/mean": 0.6704203371818249, "eval_rewards/total_composite/std": 0.37348280388575333, "eval_runtime": 90.6979, "eval_samples_per_second": 1.147, "eval_sampling/importance_sampling_ratio/max": 1.352581189228938, "eval_sampling/importance_sampling_ratio/mean": 1.002858510384193, "eval_sampling/importance_sampling_ratio/min": 0.45446773446523225, "eval_sampling/sampling_logp_difference/max": 0.8244214149621817, "eval_sampling/sampling_logp_difference/mean": 0.011250602045597939, "eval_steps_per_second": 0.143, "step": 450 }, { "clip_ratio/high_max": 0.0517552737146616, "clip_ratio/high_mean": 0.0517552737146616, "clip_ratio/low_mean": 0.03685897495597601, "clip_ratio/low_min": 0.03685897495597601, "clip_ratio/region_mean": 0.08861424867063761, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 60.25, "completions/mean_terminated_length": 60.25, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.7900894470512867, "epoch": 0.017415817114612295, "frac_reward_zero_std": 0.0, "grad_norm": 8.833879470825195, "learning_rate": 8.636363636363637e-06, "loss": -0.0468, "num_tokens": 969993.0, "reward": 0.6889505386352539, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6889505386352539, "reward_meter_std": 0.3999699652194977, "reward_std": 0.3999699354171753, "reward_total_composite_mean": 0.6889505386352539, "reward_total_composite_std": 0.3999699652194977, "reward_total_mean": 0.6889505386352539, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6889505386352539, "rewards/meter/std": 0.3999699652194977, "rewards/total_composite/mean": 0.6889505386352539, "rewards/total_composite/std": 0.3999699652194977, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0142425298690796, "sampling/importance_sampling_ratio/min": 0.35178372263908386, "sampling/sampling_logp_difference/max": 1.3110675811767578, "sampling/sampling_logp_difference/mean": 0.09507620334625244, "step": 451 }, { "clip_ratio/high_max": 0.02525400766171515, "clip_ratio/high_mean": 0.02525400766171515, "clip_ratio/low_mean": 0.006992337410338223, "clip_ratio/low_min": 0.006992337410338223, "clip_ratio/region_mean": 0.03224634507205337, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 94.0, "completions/mean_terminated_length": 94.0, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.20060661621391773, "epoch": 0.01745443311708372, "frac_reward_zero_std": 0.0, "grad_norm": 4.048279762268066, "learning_rate": 8.633333333333334e-06, "loss": -0.0294, "num_tokens": 972209.0, "reward": 0.9136518239974976, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9136518239974976, "reward_meter_std": 0.0902651846408844, "reward_std": 0.0902651846408844, "reward_total_composite_mean": 0.9136518239974976, "reward_total_composite_std": 0.0902651846408844, "reward_total_mean": 0.9136518239974976, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9136518239974976, "rewards/meter/std": 0.0902651846408844, "rewards/total_composite/mean": 0.9136518239974976, "rewards/total_composite/std": 0.0902651846408844, "sampling/importance_sampling_ratio/max": 1.876058578491211, "sampling/importance_sampling_ratio/mean": 1.0003708600997925, "sampling/importance_sampling_ratio/min": 0.3134796619415283, "sampling/sampling_logp_difference/max": 1.1600208282470703, "sampling/sampling_logp_difference/mean": 0.03461015596985817, "step": 452 }, { "clip_ratio/high_max": 0.015870221075601876, "clip_ratio/high_mean": 0.015870221075601876, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.015870221075601876, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 70.125, "completions/mean_terminated_length": 70.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.08665410289540887, "epoch": 0.017493049119555144, "frac_reward_zero_std": 0.0, "grad_norm": 6.656115531921387, "learning_rate": 8.630303030303032e-06, "loss": 0.0093, "num_tokens": 973962.0, "reward": 0.9888208508491516, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9888208508491516, "reward_meter_std": 0.006749412976205349, "reward_std": 0.006749419495463371, "reward_total_composite_mean": 0.9888208508491516, "reward_total_composite_std": 0.006749412976205349, "reward_total_mean": 0.9888208508491516, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9888208508491516, "rewards/meter/std": 0.006749412976205349, "rewards/total_composite/mean": 0.9888208508491516, "rewards/total_composite/std": 0.006749412976205349, "sampling/importance_sampling_ratio/max": 1.4743931293487549, "sampling/importance_sampling_ratio/mean": 0.9998013377189636, "sampling/importance_sampling_ratio/min": 0.22037020325660706, "sampling/sampling_logp_difference/max": 1.512446403503418, "sampling/sampling_logp_difference/mean": 0.02052040584385395, "step": 453 }, { "clip_ratio/high_max": 0.0574263078160584, "clip_ratio/high_mean": 0.0574263078160584, "clip_ratio/low_mean": 0.01666666753590107, "clip_ratio/low_min": 0.01666666753590107, "clip_ratio/region_mean": 0.07409297535195947, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 58.625, "completions/mean_terminated_length": 58.625, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.6852592751383781, "epoch": 0.017531665122026568, "frac_reward_zero_std": 0.0, "grad_norm": 9.838415145874023, "learning_rate": 8.627272727272727e-06, "loss": -0.0522, "num_tokens": 975703.0, "reward": 0.9708679914474487, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9708679914474487, "reward_meter_std": 0.03722724691033363, "reward_std": 0.037227239459753036, "reward_total_composite_mean": 0.9708679914474487, "reward_total_composite_std": 0.03722724691033363, "reward_total_mean": 0.9708679914474487, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9708679914474487, "rewards/meter/std": 0.03722724691033363, "rewards/total_composite/mean": 0.9708679914474487, "rewards/total_composite/std": 0.03722724691033363, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0130358934402466, "sampling/importance_sampling_ratio/min": 0.18911553919315338, "sampling/sampling_logp_difference/max": 1.6653971672058105, "sampling/sampling_logp_difference/mean": 0.07885369658470154, "step": 454 }, { "clip_ratio/high_max": 0.03188905236311257, "clip_ratio/high_mean": 0.03188905236311257, "clip_ratio/low_mean": 0.026715174899436533, "clip_ratio/low_min": 0.026715174899436533, "clip_ratio/region_mean": 0.0586042272625491, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 62.5, "completions/mean_terminated_length": 62.5, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.5540340095758438, "epoch": 0.017570281124497992, "frac_reward_zero_std": 0.0, "grad_norm": 7.684765815734863, "learning_rate": 8.624242424242424e-06, "loss": 0.0665, "num_tokens": 977683.0, "reward": 0.6929647922515869, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6929647922515869, "reward_meter_std": 0.3059764504432678, "reward_std": 0.3059764504432678, "reward_total_composite_mean": 0.6929647922515869, "reward_total_composite_std": 0.3059764504432678, "reward_total_mean": 0.6929647922515869, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6929647922515869, "rewards/meter/std": 0.3059764504432678, "rewards/total_composite/mean": 0.6929647922515869, "rewards/total_composite/std": 0.3059764504432678, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0214698314666748, "sampling/importance_sampling_ratio/min": 0.06062639132142067, "sampling/sampling_logp_difference/max": 2.803025007247925, "sampling/sampling_logp_difference/mean": 0.07383247464895248, "step": 455 }, { "clip_ratio/high_max": 0.037544333608821034, "clip_ratio/high_mean": 0.037544333608821034, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/region_mean": 0.042544333497062325, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 75.625, "completions/mean_terminated_length": 75.625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.3263458888977766, "epoch": 0.017608897126969416, "frac_reward_zero_std": 0.0, "grad_norm": 7.524571895599365, "learning_rate": 8.621212121212122e-06, "loss": 0.0137, "num_tokens": 979768.0, "reward": 0.8802084922790527, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8802084922790527, "reward_meter_std": 0.23987267911434174, "reward_std": 0.23987266421318054, "reward_total_composite_mean": 0.8802084922790527, "reward_total_composite_std": 0.23987267911434174, "reward_total_mean": 0.8802084922790527, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8802084922790527, "rewards/meter/std": 0.23987267911434174, "rewards/total_composite/mean": 0.8802084922790527, "rewards/total_composite/std": 0.23987267911434174, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0096123218536377, "sampling/importance_sampling_ratio/min": 0.1034177839756012, "sampling/sampling_logp_difference/max": 2.2689783573150635, "sampling/sampling_logp_difference/mean": 0.0491391159594059, "step": 456 }, { "clip_ratio/high_max": 0.018728531897068024, "clip_ratio/high_mean": 0.018728531897068024, "clip_ratio/low_mean": 0.026498167659156024, "clip_ratio/low_min": 0.026498167659156024, "clip_ratio/region_mean": 0.04522669955622405, "completions/clipped_ratio": 0.0, "completions/max_length": 130.0, "completions/max_terminated_length": 130.0, "completions/mean_length": 110.625, "completions/mean_terminated_length": 110.625, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.3454656656831503, "epoch": 0.01764751312944084, "frac_reward_zero_std": 0.0, "grad_norm": 6.170909404754639, "learning_rate": 8.618181818181819e-06, "loss": 0.0947, "num_tokens": 982005.0, "reward": 0.3757762312889099, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.37684983015060425, "reward_meter_std": 0.48171141743659973, "reward_std": 0.4826580584049225, "reward_total_composite_mean": 0.3757762312889099, "reward_total_composite_std": 0.4826580584049225, "reward_total_mean": 0.3757762312889099, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.37684983015060425, "rewards/meter/std": 0.48171141743659973, "rewards/total_composite/mean": 0.3757762312889099, "rewards/total_composite/std": 0.4826580584049225, "sampling/importance_sampling_ratio/max": 1.8442530632019043, "sampling/importance_sampling_ratio/mean": 1.0049611330032349, "sampling/importance_sampling_ratio/min": 0.22940349578857422, "sampling/sampling_logp_difference/max": 1.4722728729248047, "sampling/sampling_logp_difference/mean": 0.0551149956882, "step": 457 }, { "clip_ratio/high_max": 0.07460826355963945, "clip_ratio/high_mean": 0.07460826355963945, "clip_ratio/low_mean": 0.04874078743159771, "clip_ratio/low_min": 0.04874078743159771, "clip_ratio/region_mean": 0.12334905099123716, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 25.125, "completions/mean_terminated_length": 25.125, "completions/min_length": 10.0, "completions/min_terminated_length": 10.0, "entropy": 1.231294609606266, "epoch": 0.017686129131912264, "frac_reward_zero_std": 0.0, "grad_norm": 16.211706161499023, "learning_rate": 8.615151515151516e-06, "loss": -0.0317, "num_tokens": 983398.0, "reward": 0.6279522180557251, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6279522180557251, "reward_meter_std": 0.4916090965270996, "reward_std": 0.491609126329422, "reward_total_composite_mean": 0.6279522180557251, "reward_total_composite_std": 0.4916090965270996, "reward_total_mean": 0.6279522180557251, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6279522180557251, "rewards/meter/std": 0.4916090965270996, "rewards/total_composite/mean": 0.6279522180557251, "rewards/total_composite/std": 0.4916090965270996, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0198932886123657, "sampling/importance_sampling_ratio/min": 0.29847922921180725, "sampling/sampling_logp_difference/max": 1.209054946899414, "sampling/sampling_logp_difference/mean": 0.1369742900133133, "step": 458 }, { "clip_ratio/high_max": 0.04640375077724457, "clip_ratio/high_mean": 0.04640375077724457, "clip_ratio/low_mean": 0.02632259437814355, "clip_ratio/low_min": 0.02632259437814355, "clip_ratio/region_mean": 0.07272634515538812, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 74.875, "completions/mean_terminated_length": 74.875, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.4475377518683672, "epoch": 0.01772474513438369, "frac_reward_zero_std": 0.0, "grad_norm": 7.541397571563721, "learning_rate": 8.612121212121213e-06, "loss": 0.1012, "num_tokens": 985277.0, "reward": 0.500255286693573, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.500255286693573, "reward_meter_std": 0.3483890891075134, "reward_std": 0.3483890891075134, "reward_total_composite_mean": 0.500255286693573, "reward_total_composite_std": 0.3483890891075134, "reward_total_mean": 0.500255286693573, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.500255286693573, "rewards/meter/std": 0.3483890891075134, "rewards/total_composite/mean": 0.500255286693573, "rewards/total_composite/std": 0.3483890891075134, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000501036643982, "sampling/importance_sampling_ratio/min": 0.2836383879184723, "sampling/sampling_logp_difference/max": 1.2600550651550293, "sampling/sampling_logp_difference/mean": 0.08134466409683228, "step": 459 }, { "clip_ratio/high_max": 0.012001242313999683, "clip_ratio/high_mean": 0.012001242313999683, "clip_ratio/low_mean": 0.017407751642167568, "clip_ratio/low_min": 0.017407751642167568, "clip_ratio/region_mean": 0.02940899395616725, "completions/clipped_ratio": 0.0, "completions/max_length": 154.0, "completions/max_terminated_length": 154.0, "completions/mean_length": 136.375, "completions/mean_terminated_length": 136.375, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.2906641475856304, "epoch": 0.017763361136855112, "frac_reward_zero_std": 0.0, "grad_norm": 4.076127529144287, "learning_rate": 8.60909090909091e-06, "loss": -0.0349, "num_tokens": 987752.0, "reward": 0.5672988295555115, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5672988295555115, "reward_meter_std": 0.4312629699707031, "reward_std": 0.4312629699707031, "reward_total_composite_mean": 0.5672988295555115, "reward_total_composite_std": 0.4312629699707031, "reward_total_mean": 0.5672988295555115, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5672988295555115, "rewards/meter/std": 0.4312629699707031, "rewards/total_composite/mean": 0.5672988295555115, "rewards/total_composite/std": 0.4312629699707031, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0044218301773071, "sampling/importance_sampling_ratio/min": 0.1266932338476181, "sampling/sampling_logp_difference/max": 2.0659866333007812, "sampling/sampling_logp_difference/mean": 0.045063383877277374, "step": 460 }, { "clip_ratio/high_max": 0.009281257749535143, "clip_ratio/high_mean": 0.009281257749535143, "clip_ratio/low_mean": 0.002057613106444478, "clip_ratio/low_min": 0.002057613106444478, "clip_ratio/region_mean": 0.011338870855979621, "completions/clipped_ratio": 0.0, "completions/max_length": 243.0, "completions/max_terminated_length": 243.0, "completions/mean_length": 208.75, "completions/mean_terminated_length": 208.75, "completions/min_length": 192.0, "completions/min_terminated_length": 192.0, "entropy": 0.08123606955632567, "epoch": 0.017801977139326536, "frac_reward_zero_std": 0.0, "grad_norm": 2.037872552871704, "learning_rate": 8.606060606060606e-06, "loss": 0.059, "num_tokens": 991070.0, "reward": 0.9722763299942017, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.9929186105728149, "reward_meter_std": 0.0022024763748049736, "reward_std": 0.059263937175273895, "reward_total_composite_mean": 0.9722763299942017, "reward_total_composite_std": 0.0592639334499836, "reward_total_mean": 0.9722763299942017, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.9929186105728149, "rewards/meter/std": 0.0022024763748049736, "rewards/total_composite/mean": 0.9722763299942017, "rewards/total_composite/std": 0.0592639334499836, "sampling/importance_sampling_ratio/max": 1.966139793395996, "sampling/importance_sampling_ratio/mean": 1.0022977590560913, "sampling/importance_sampling_ratio/min": 0.16790510714054108, "sampling/sampling_logp_difference/max": 1.7843563556671143, "sampling/sampling_logp_difference/mean": 0.014968657866120338, "step": 461 }, { "clip_ratio/high_max": 0.05066513631027192, "clip_ratio/high_mean": 0.05066513631027192, "clip_ratio/low_mean": 0.012249511666595936, "clip_ratio/low_min": 0.012249511666595936, "clip_ratio/region_mean": 0.06291464797686785, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 87.375, "completions/mean_terminated_length": 87.375, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.557247344404459, "epoch": 0.01784059314179796, "frac_reward_zero_std": 0.0, "grad_norm": 6.325527667999268, "learning_rate": 8.603030303030303e-06, "loss": 0.0298, "num_tokens": 993009.0, "reward": 0.82567298412323, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.82567298412323, "reward_meter_std": 0.2936802804470062, "reward_std": 0.29368025064468384, "reward_total_composite_mean": 0.82567298412323, "reward_total_composite_std": 0.2936802804470062, "reward_total_mean": 0.82567298412323, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.82567298412323, "rewards/meter/std": 0.2936802804470062, "rewards/total_composite/mean": 0.82567298412323, "rewards/total_composite/std": 0.2936802804470062, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.012847661972046, "sampling/importance_sampling_ratio/min": 0.34010180830955505, "sampling/sampling_logp_difference/max": 1.0785102844238281, "sampling/sampling_logp_difference/mean": 0.06957826763391495, "step": 462 }, { "clip_ratio/high_max": 0.02107094577513635, "clip_ratio/high_mean": 0.02107094577513635, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.024858824675902724, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.1741385543718934, "epoch": 0.017879209144269385, "frac_reward_zero_std": 0.0, "grad_norm": 5.930828094482422, "learning_rate": 8.6e-06, "loss": 0.0017, "num_tokens": 994937.0, "reward": 0.9922396540641785, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9922396540641785, "reward_meter_std": 0.002804464427754283, "reward_std": 0.0028044627979397774, "reward_total_composite_mean": 0.9922396540641785, "reward_total_composite_std": 0.002804464427754283, "reward_total_mean": 0.9922396540641785, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9922396540641785, "rewards/meter/std": 0.002804464427754283, "rewards/total_composite/mean": 0.9922396540641785, "rewards/total_composite/std": 0.002804464427754283, "sampling/importance_sampling_ratio/max": 1.8714525699615479, "sampling/importance_sampling_ratio/mean": 1.003735065460205, "sampling/importance_sampling_ratio/min": 0.3877050578594208, "sampling/sampling_logp_difference/max": 0.9475104808807373, "sampling/sampling_logp_difference/mean": 0.033212851732969284, "step": 463 }, { "clip_ratio/high_max": 0.08965541888028383, "clip_ratio/high_mean": 0.08965541888028383, "clip_ratio/low_mean": 0.008184524020180106, "clip_ratio/low_min": 0.008184524020180106, "clip_ratio/region_mean": 0.09783994290046394, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 48.0, "completions/mean_length": 155.25, "completions/mean_terminated_length": 36.333335876464844, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 1.133492972701788, "epoch": 0.01791782514674081, "frac_reward_zero_std": 0.0, "grad_norm": 2.3135251998901367, "learning_rate": 8.596969696969698e-06, "loss": 0.0244, "num_tokens": 996427.0, "reward": 0.6161172389984131, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6161172389984131, "reward_meter_std": 0.4752645790576935, "reward_std": 0.47526460886001587, "reward_total_composite_mean": 0.6161172389984131, "reward_total_composite_std": 0.4752645790576935, "reward_total_mean": 0.6161172389984131, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6161172389984131, "rewards/meter/std": 0.4752645790576935, "rewards/total_composite/mean": 0.6161172389984131, "rewards/total_composite/std": 0.4752645790576935, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9996461868286133, "sampling/importance_sampling_ratio/min": 0.307363361120224, "sampling/sampling_logp_difference/max": 1.1797246932983398, "sampling/sampling_logp_difference/mean": 0.12042637169361115, "step": 464 }, { "clip_ratio/high_max": 0.017931917682290077, "clip_ratio/high_mean": 0.017931917682290077, "clip_ratio/low_mean": 0.015818335115909576, "clip_ratio/low_min": 0.015818335115909576, "clip_ratio/region_mean": 0.033750252798199654, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 120.875, "completions/mean_terminated_length": 120.875, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.2514910716563463, "epoch": 0.017956441149212233, "frac_reward_zero_std": 0.0, "grad_norm": 4.309906959533691, "learning_rate": 8.593939393939395e-06, "loss": 0.0135, "num_tokens": 998754.0, "reward": 0.6896439790725708, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8134843111038208, "reward_meter_std": 0.3482135534286499, "reward_std": 0.44019806385040283, "reward_total_composite_mean": 0.6896439790725708, "reward_total_composite_std": 0.4401980936527252, "reward_total_mean": 0.6896439790725708, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8134843111038208, "rewards/meter/std": 0.3482135534286499, "rewards/total_composite/mean": 0.6896439790725708, "rewards/total_composite/std": 0.4401980936527252, "sampling/importance_sampling_ratio/max": 1.797261357307434, "sampling/importance_sampling_ratio/mean": 1.0006272792816162, "sampling/importance_sampling_ratio/min": 0.311477929353714, "sampling/sampling_logp_difference/max": 1.1664267778396606, "sampling/sampling_logp_difference/mean": 0.03921005502343178, "step": 465 }, { "clip_ratio/high_max": 0.03716489998623729, "clip_ratio/high_mean": 0.03716489998623729, "clip_ratio/low_mean": 0.006340579595416784, "clip_ratio/low_min": 0.006340579595416784, "clip_ratio/region_mean": 0.04350547958165407, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 120.125, "completions/mean_terminated_length": 120.125, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.291955066844821, "epoch": 0.017995057151683657, "frac_reward_zero_std": 0.0, "grad_norm": 5.061103820800781, "learning_rate": 8.590909090909092e-06, "loss": 0.0569, "num_tokens": 1000955.0, "reward": 0.9295108914375305, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9295108914375305, "reward_meter_std": 0.18052063882350922, "reward_std": 0.18052060902118683, "reward_total_composite_mean": 0.9295108914375305, "reward_total_composite_std": 0.18052063882350922, "reward_total_mean": 0.9295108914375305, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9295108914375305, "rewards/meter/std": 0.18052063882350922, "rewards/total_composite/mean": 0.9295108914375305, "rewards/total_composite/std": 0.18052063882350922, "sampling/importance_sampling_ratio/max": 1.9984016418457031, "sampling/importance_sampling_ratio/mean": 1.0039215087890625, "sampling/importance_sampling_ratio/min": 0.09835071116685867, "sampling/sampling_logp_difference/max": 2.3192155361175537, "sampling/sampling_logp_difference/mean": 0.04497542977333069, "step": 466 }, { "clip_ratio/high_max": 0.008227241458371282, "clip_ratio/high_mean": 0.008227241458371282, "clip_ratio/low_mean": 0.00698582676704973, "clip_ratio/low_min": 0.00698582676704973, "clip_ratio/region_mean": 0.015213068225421011, "completions/clipped_ratio": 0.0, "completions/max_length": 174.0, "completions/max_terminated_length": 174.0, "completions/mean_length": 152.0, "completions/mean_terminated_length": 152.0, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "entropy": 0.0712776998989284, "epoch": 0.01803367315415508, "frac_reward_zero_std": 0.0, "grad_norm": 2.9019546508789062, "learning_rate": 8.587878787878788e-06, "loss": 0.0437, "num_tokens": 1003675.0, "reward": 0.8826419711112976, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8826419711112976, "reward_meter_std": 0.06341774016618729, "reward_std": 0.06341774016618729, "reward_total_composite_mean": 0.8826419711112976, "reward_total_composite_std": 0.06341774016618729, "reward_total_mean": 0.8826419711112976, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8826419711112976, "rewards/meter/std": 0.06341774016618729, "rewards/total_composite/mean": 0.8826419711112976, "rewards/total_composite/std": 0.06341774016618729, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0029544830322266, "sampling/importance_sampling_ratio/min": 0.26845496892929077, "sampling/sampling_logp_difference/max": 1.3150720596313477, "sampling/sampling_logp_difference/mean": 0.015705382451415062, "step": 467 }, { "clip_ratio/high_max": 0.0006329113966785371, "clip_ratio/high_mean": 0.0006329113966785371, "clip_ratio/low_mean": 0.0020095510117243975, "clip_ratio/low_min": 0.0020095510117243975, "clip_ratio/region_mean": 0.0026424624084029347, "completions/clipped_ratio": 0.0, "completions/max_length": 413.0, "completions/max_terminated_length": 413.0, "completions/mean_length": 386.25, "completions/mean_terminated_length": 386.25, "completions/min_length": 359.0, "completions/min_terminated_length": 359.0, "entropy": 0.028902734396979213, "epoch": 0.018072289156626505, "frac_reward_zero_std": 0.0, "grad_norm": 0.9965296983718872, "learning_rate": 8.584848484848485e-06, "loss": -0.0241, "num_tokens": 1008541.0, "reward": 0.906877875328064, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9431818723678589, "reward_count_adherence_std": 0.047049909830093384, "reward_meter_mean": 0.9619865417480469, "reward_meter_std": 0.025031376630067825, "reward_std": 0.04043539986014366, "reward_total_composite_mean": 0.906877875328064, "reward_total_composite_std": 0.04043539986014366, "reward_total_mean": 0.906877875328064, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9431818723678589, "rewards/count_adherence/std": 0.047049909830093384, "rewards/meter/mean": 0.9619865417480469, "rewards/meter/std": 0.025031376630067825, "rewards/total_composite/mean": 0.906877875328064, "rewards/total_composite/std": 0.04043539986014366, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0018101930618286, "sampling/importance_sampling_ratio/min": 0.4794699549674988, "sampling/sampling_logp_difference/max": 1.920891523361206, "sampling/sampling_logp_difference/mean": 0.00543476827442646, "step": 468 }, { "clip_ratio/high_max": 0.026214548386633396, "clip_ratio/high_mean": 0.026214548386633396, "clip_ratio/low_mean": 0.009848485235124826, "clip_ratio/low_min": 0.009848485235124826, "clip_ratio/region_mean": 0.03606303362175822, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 37.0, "completions/mean_terminated_length": 37.0, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.17447292152792215, "epoch": 0.01811090515909793, "frac_reward_zero_std": 0.0, "grad_norm": 7.141157150268555, "learning_rate": 8.581818181818183e-06, "loss": 0.1817, "num_tokens": 1010085.0, "reward": 0.7328898906707764, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.857481062412262, "reward_meter_std": 0.2525491416454315, "reward_std": 0.38510987162590027, "reward_total_composite_mean": 0.7328898906707764, "reward_total_composite_std": 0.38510990142822266, "reward_total_mean": 0.7328898906707764, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.857481062412262, "rewards/meter/std": 0.2525491416454315, "rewards/total_composite/mean": 0.7328898906707764, "rewards/total_composite/std": 0.38510990142822266, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9989219903945923, "sampling/importance_sampling_ratio/min": 0.2123665064573288, "sampling/sampling_logp_difference/max": 1.5494416952133179, "sampling/sampling_logp_difference/mean": 0.04425249993801117, "step": 469 }, { "clip_ratio/high_max": 0.0195100650889799, "clip_ratio/high_mean": 0.0195100650889799, "clip_ratio/low_mean": 0.007512626354582608, "clip_ratio/low_min": 0.007512626354582608, "clip_ratio/region_mean": 0.027022691443562508, "completions/clipped_ratio": 0.0, "completions/max_length": 100.0, "completions/max_terminated_length": 100.0, "completions/mean_length": 96.375, "completions/mean_terminated_length": 96.375, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.22278593480587006, "epoch": 0.018149521161569353, "frac_reward_zero_std": 0.0, "grad_norm": 7.436746597290039, "learning_rate": 8.57878787878788e-06, "loss": 0.0294, "num_tokens": 1012136.0, "reward": 0.9915547370910645, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9915547370910645, "reward_meter_std": 0.0038915486074984074, "reward_std": 0.0038915553595870733, "reward_total_composite_mean": 0.9915547370910645, "reward_total_composite_std": 0.0038915486074984074, "reward_total_mean": 0.9915547370910645, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9915547370910645, "rewards/meter/std": 0.0038915486074984074, "rewards/total_composite/mean": 0.9915547370910645, "rewards/total_composite/std": 0.0038915486074984074, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0057390928268433, "sampling/importance_sampling_ratio/min": 0.24664175510406494, "sampling/sampling_logp_difference/max": 1.3998184204101562, "sampling/sampling_logp_difference/mean": 0.038666851818561554, "step": 470 }, { "clip_ratio/high_max": 0.09589226730167866, "clip_ratio/high_mean": 0.09589226730167866, "clip_ratio/low_mean": 0.07797870878130198, "clip_ratio/low_min": 0.07797870878130198, "clip_ratio/region_mean": 0.17387097608298063, "completions/clipped_ratio": 0.0, "completions/max_length": 47.0, "completions/max_terminated_length": 47.0, "completions/mean_length": 40.5, "completions/mean_terminated_length": 40.5, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 1.0023160837590694, "epoch": 0.018188137164040778, "frac_reward_zero_std": 0.0, "grad_norm": 15.974438667297363, "learning_rate": 8.575757575757575e-06, "loss": 0.0586, "num_tokens": 1013700.0, "reward": 0.38729679584503174, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.38729679584503174, "reward_meter_std": 0.3866852819919586, "reward_std": 0.3866852819919586, "reward_total_composite_mean": 0.38729679584503174, "reward_total_composite_std": 0.3866852819919586, "reward_total_mean": 0.38729679584503174, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.38729679584503174, "rewards/meter/std": 0.3866852819919586, "rewards/total_composite/mean": 0.38729679584503174, "rewards/total_composite/std": 0.3866852819919586, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0078424215316772, "sampling/importance_sampling_ratio/min": 0.1371387541294098, "sampling/sampling_logp_difference/max": 1.9867620468139648, "sampling/sampling_logp_difference/mean": 0.16885001957416534, "step": 471 }, { "clip_ratio/high_max": 0.004306173766963184, "clip_ratio/high_mean": 0.004306173766963184, "clip_ratio/low_mean": 0.004235844942741096, "clip_ratio/low_min": 0.004235844942741096, "clip_ratio/region_mean": 0.00854201870970428, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 278.75, "completions/mean_terminated_length": 278.75, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "entropy": 0.03723532357253134, "epoch": 0.0182267531665122, "frac_reward_zero_std": 0.0, "grad_norm": 1.0963542461395264, "learning_rate": 8.572727272727274e-06, "loss": -0.0246, "num_tokens": 1017570.0, "reward": 0.8407608866691589, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.924202561378479, "reward_meter_std": 0.0937681496143341, "reward_std": 0.10190220922231674, "reward_total_composite_mean": 0.8407608866691589, "reward_total_composite_std": 0.10190222412347794, "reward_total_mean": 0.8407608866691589, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.924202561378479, "rewards/meter/std": 0.0937681496143341, "rewards/total_composite/mean": 0.8407608866691589, "rewards/total_composite/std": 0.10190222412347794, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000677227973938, "sampling/importance_sampling_ratio/min": 0.370554655790329, "sampling/sampling_logp_difference/max": 0.992754340171814, "sampling/sampling_logp_difference/mean": 0.008891636505723, "step": 472 }, { "clip_ratio/high_max": 0.01820858521386981, "clip_ratio/high_mean": 0.01820858521386981, "clip_ratio/low_mean": 0.01952884637285024, "clip_ratio/low_min": 0.01952884637285024, "clip_ratio/region_mean": 0.03773743158672005, "completions/clipped_ratio": 0.0, "completions/max_length": 196.0, "completions/max_terminated_length": 196.0, "completions/mean_length": 171.25, "completions/mean_terminated_length": 171.25, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.17230255994945765, "epoch": 0.018265369168983626, "frac_reward_zero_std": 0.0, "grad_norm": 5.147512912750244, "learning_rate": 8.56969696969697e-06, "loss": -0.0118, "num_tokens": 1020348.0, "reward": 0.49889272451400757, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_meter_mean": 0.5626472234725952, "reward_meter_std": 0.38277679681777954, "reward_std": 0.33463039994239807, "reward_total_composite_mean": 0.49889272451400757, "reward_total_composite_std": 0.33463042974472046, "reward_total_mean": 0.49889272451400757, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/meter/mean": 0.5626472234725952, "rewards/meter/std": 0.38277679681777954, "rewards/total_composite/mean": 0.49889272451400757, "rewards/total_composite/std": 0.33463042974472046, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9987762570381165, "sampling/importance_sampling_ratio/min": 0.12169203162193298, "sampling/sampling_logp_difference/max": 2.106261730194092, "sampling/sampling_logp_difference/mean": 0.053236186504364014, "step": 473 }, { "clip_ratio/high_max": 0.004681251273723319, "clip_ratio/high_mean": 0.004681251273723319, "clip_ratio/low_mean": 0.005101353977806866, "clip_ratio/low_min": 0.005101353977806866, "clip_ratio/region_mean": 0.009782605251530185, "completions/clipped_ratio": 0.0, "completions/max_length": 291.0, "completions/max_terminated_length": 291.0, "completions/mean_length": 253.625, "completions/mean_terminated_length": 253.625, "completions/min_length": 223.0, "completions/min_terminated_length": 223.0, "entropy": 0.10501971561461687, "epoch": 0.01830398517145505, "frac_reward_zero_std": 0.0, "grad_norm": 2.0184075832366943, "learning_rate": 8.566666666666667e-06, "loss": 0.0284, "num_tokens": 1024089.0, "reward": 0.915224552154541, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_meter_mean": 0.998430073261261, "reward_meter_std": 0.0012139956234022975, "reward_std": 0.08891873806715012, "reward_total_composite_mean": 0.915224552154541, "reward_total_composite_std": 0.08891873806715012, "reward_total_mean": 0.915224552154541, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/meter/mean": 0.998430073261261, "rewards/meter/std": 0.0012139956234022975, "rewards/total_composite/mean": 0.915224552154541, "rewards/total_composite/std": 0.08891873806715012, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9997615814208984, "sampling/importance_sampling_ratio/min": 0.1190701276063919, "sampling/sampling_logp_difference/max": 2.128042697906494, "sampling/sampling_logp_difference/mean": 0.020224176347255707, "step": 474 }, { "clip_ratio/high_max": 0.04206577688455582, "clip_ratio/high_mean": 0.04206577688455582, "clip_ratio/low_mean": 0.006897549610584974, "clip_ratio/low_min": 0.006897549610584974, "clip_ratio/region_mean": 0.04896332649514079, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 71.25, "completions/mean_terminated_length": 71.25, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2513603400439024, "epoch": 0.018342601173926474, "frac_reward_zero_std": 0.0, "grad_norm": 5.0996994972229, "learning_rate": 8.563636363636364e-06, "loss": 0.0063, "num_tokens": 1025987.0, "reward": 0.8450901508331299, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8450901508331299, "reward_meter_std": 0.1726076304912567, "reward_std": 0.17260761559009552, "reward_total_composite_mean": 0.8450901508331299, "reward_total_composite_std": 0.1726076304912567, "reward_total_mean": 0.8450901508331299, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8450901508331299, "rewards/meter/std": 0.1726076304912567, "rewards/total_composite/mean": 0.8450901508331299, "rewards/total_composite/std": 0.1726076304912567, "sampling/importance_sampling_ratio/max": 1.6401515007019043, "sampling/importance_sampling_ratio/mean": 0.9991965889930725, "sampling/importance_sampling_ratio/min": 0.3256930410861969, "sampling/sampling_logp_difference/max": 1.1217999458312988, "sampling/sampling_logp_difference/mean": 0.043718110769987106, "step": 475 }, { "clip_ratio/high_max": 0.012941297609359026, "clip_ratio/high_mean": 0.012941297609359026, "clip_ratio/low_mean": 0.009093915577977896, "clip_ratio/low_min": 0.009093915577977896, "clip_ratio/region_mean": 0.02203521318733692, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 29.875, "completions/mean_terminated_length": 29.875, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.20345518365502357, "epoch": 0.018381217176397898, "frac_reward_zero_std": 0.0, "grad_norm": 7.3300299644470215, "learning_rate": 8.560606060606062e-06, "loss": -0.0343, "num_tokens": 1027490.0, "reward": 0.9858591556549072, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9858591556549072, "reward_meter_std": 0.004257791675627232, "reward_std": 0.004257792141288519, "reward_total_composite_mean": 0.9858591556549072, "reward_total_composite_std": 0.004257791675627232, "reward_total_mean": 0.9858591556549072, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9858591556549072, "rewards/meter/std": 0.004257791675627232, "rewards/total_composite/mean": 0.9858591556549072, "rewards/total_composite/std": 0.004257791675627232, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0109765529632568, "sampling/importance_sampling_ratio/min": 0.5488837957382202, "sampling/sampling_logp_difference/max": 0.8543341159820557, "sampling/sampling_logp_difference/mean": 0.03374743461608887, "step": 476 }, { "clip_ratio/high_max": 0.011585089145228267, "clip_ratio/high_mean": 0.011585089145228267, "clip_ratio/low_mean": 0.005435622064396739, "clip_ratio/low_min": 0.005435622064396739, "clip_ratio/region_mean": 0.017020711209625006, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 108.5, "completions/mean_terminated_length": 108.5, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.15065084025263786, "epoch": 0.018419833178869322, "frac_reward_zero_std": 0.0, "grad_norm": 3.7667415142059326, "learning_rate": 8.557575757575757e-06, "loss": 0.018, "num_tokens": 1029622.0, "reward": 0.9522513151168823, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9522513151168823, "reward_meter_std": 0.056059498339891434, "reward_std": 0.05605950206518173, "reward_total_composite_mean": 0.9522513151168823, "reward_total_composite_std": 0.056059498339891434, "reward_total_mean": 0.9522513151168823, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9522513151168823, "rewards/meter/std": 0.056059498339891434, "rewards/total_composite/mean": 0.9522513151168823, "rewards/total_composite/std": 0.056059498339891434, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0057148933410645, "sampling/importance_sampling_ratio/min": 0.3075287938117981, "sampling/sampling_logp_difference/max": 1.1791865825653076, "sampling/sampling_logp_difference/mean": 0.026357771828770638, "step": 477 }, { "clip_ratio/high_max": 0.04250439163297415, "clip_ratio/high_mean": 0.04250439163297415, "clip_ratio/low_mean": 0.02712087077088654, "clip_ratio/low_min": 0.02712087077088654, "clip_ratio/region_mean": 0.06962526240386069, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 37.875, "completions/mean_terminated_length": 37.875, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.5912935622036457, "epoch": 0.018458449181340746, "frac_reward_zero_std": 0.0, "grad_norm": 10.037378311157227, "learning_rate": 8.554545454545456e-06, "loss": -0.0113, "num_tokens": 1031189.0, "reward": 0.978717565536499, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.978717565536499, "reward_meter_std": 0.019986465573310852, "reward_std": 0.019986478611826897, "reward_total_composite_mean": 0.978717565536499, "reward_total_composite_std": 0.019986465573310852, "reward_total_mean": 0.978717565536499, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.978717565536499, "rewards/meter/std": 0.019986465573310852, "rewards/total_composite/mean": 0.978717565536499, "rewards/total_composite/std": 0.019986465573310852, "sampling/importance_sampling_ratio/max": 1.5463435649871826, "sampling/importance_sampling_ratio/mean": 1.0022932291030884, "sampling/importance_sampling_ratio/min": 0.23787212371826172, "sampling/sampling_logp_difference/max": 1.4360220432281494, "sampling/sampling_logp_difference/mean": 0.08604246377944946, "step": 478 }, { "clip_ratio/high_max": 0.05710198846645653, "clip_ratio/high_mean": 0.05710198846645653, "clip_ratio/low_mean": 0.018391148187220097, "clip_ratio/low_min": 0.018391148187220097, "clip_ratio/region_mean": 0.07549313665367663, "completions/clipped_ratio": 0.0, "completions/max_length": 44.0, "completions/max_terminated_length": 44.0, "completions/mean_length": 39.75, "completions/mean_terminated_length": 39.75, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.5402822978794575, "epoch": 0.01849706518381217, "frac_reward_zero_std": 0.0, "grad_norm": 10.624458312988281, "learning_rate": 8.551515151515152e-06, "loss": 0.0424, "num_tokens": 1032963.0, "reward": 0.9893389940261841, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9893389940261841, "reward_meter_std": 0.015485532581806183, "reward_std": 0.015485531650483608, "reward_total_composite_mean": 0.9893389940261841, "reward_total_composite_std": 0.015485532581806183, "reward_total_mean": 0.9893389940261841, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9893389940261841, "rewards/meter/std": 0.015485532581806183, "rewards/total_composite/mean": 0.9893389940261841, "rewards/total_composite/std": 0.015485532581806183, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0154733657836914, "sampling/importance_sampling_ratio/min": 0.2684987783432007, "sampling/sampling_logp_difference/max": 1.3149088621139526, "sampling/sampling_logp_difference/mean": 0.10419061034917831, "step": 479 }, { "clip_ratio/high_max": 0.05253493972122669, "clip_ratio/high_mean": 0.05253493972122669, "clip_ratio/low_mean": 0.05937229562550783, "clip_ratio/low_min": 0.05937229562550783, "clip_ratio/region_mean": 0.11190723534673452, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 31.625, "completions/mean_terminated_length": 31.625, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 1.2414276078343391, "epoch": 0.018535681186283594, "frac_reward_zero_std": 0.0, "grad_norm": 20.665754318237305, "learning_rate": 8.548484848484849e-06, "loss": -0.0113, "num_tokens": 1034496.0, "reward": 0.9100282192230225, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9100282192230225, "reward_meter_std": 0.14061063528060913, "reward_std": 0.14061065018177032, "reward_total_composite_mean": 0.9100282192230225, "reward_total_composite_std": 0.14061063528060913, "reward_total_mean": 0.9100282192230225, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9100282192230225, "rewards/meter/std": 0.14061063528060913, "rewards/total_composite/mean": 0.9100282192230225, "rewards/total_composite/std": 0.14061063528060913, "sampling/importance_sampling_ratio/max": 1.649492859840393, "sampling/importance_sampling_ratio/mean": 1.0112000703811646, "sampling/importance_sampling_ratio/min": 0.37617847323417664, "sampling/sampling_logp_difference/max": 0.977691650390625, "sampling/sampling_logp_difference/mean": 0.13002178072929382, "step": 480 }, { "clip_ratio/high_max": 0.08672621939331293, "clip_ratio/high_mean": 0.08672621939331293, "clip_ratio/low_mean": 0.010937499813735485, "clip_ratio/low_min": 0.010937499813735485, "clip_ratio/region_mean": 0.09766371920704842, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 76.75, "completions/mean_terminated_length": 76.75, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.5781661160290241, "epoch": 0.01857429718875502, "frac_reward_zero_std": 0.0, "grad_norm": 8.993289947509766, "learning_rate": 8.545454545454546e-06, "loss": 0.024, "num_tokens": 1036446.0, "reward": 0.9631932973861694, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9631932973861694, "reward_meter_std": 0.08677735924720764, "reward_std": 0.08677736669778824, "reward_total_composite_mean": 0.9631932973861694, "reward_total_composite_std": 0.08677735924720764, "reward_total_mean": 0.9631932973861694, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9631932973861694, "rewards/meter/std": 0.08677735924720764, "rewards/total_composite/mean": 0.9631932973861694, "rewards/total_composite/std": 0.08677735924720764, "sampling/importance_sampling_ratio/max": 1.636481761932373, "sampling/importance_sampling_ratio/mean": 0.9976180195808411, "sampling/importance_sampling_ratio/min": 0.2606494724750519, "sampling/sampling_logp_difference/max": 1.344578742980957, "sampling/sampling_logp_difference/mean": 0.0779096931219101, "step": 481 }, { "clip_ratio/high_max": 0.005991354206344113, "clip_ratio/high_mean": 0.005991354206344113, "clip_ratio/low_mean": 0.002808988792821765, "clip_ratio/low_min": 0.002808988792821765, "clip_ratio/region_mean": 0.008800342999165878, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 300.125, "completions/mean_terminated_length": 269.8571472167969, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.07355018192902207, "epoch": 0.018612913191226443, "frac_reward_zero_std": 0.0, "grad_norm": 1.2208207845687866, "learning_rate": 8.542424242424243e-06, "loss": -0.244, "num_tokens": 1040023.0, "reward": 0.7929118871688843, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.29574236273765564, "reward_meter_mean": 0.9346262216567993, "reward_meter_std": 0.16473090648651123, "reward_std": 0.3079202473163605, "reward_total_composite_mean": 0.7929118871688843, "reward_total_composite_std": 0.3079202473163605, "reward_total_mean": 0.7929118871688843, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.29574236273765564, "rewards/meter/mean": 0.9346262216567993, "rewards/meter/std": 0.16473090648651123, "rewards/total_composite/mean": 0.7929118871688843, "rewards/total_composite/std": 0.3079202473163605, "sampling/importance_sampling_ratio/max": 1.6571389436721802, "sampling/importance_sampling_ratio/mean": 1.0012937784194946, "sampling/importance_sampling_ratio/min": 0.3793216049671173, "sampling/sampling_logp_difference/max": 0.9693708419799805, "sampling/sampling_logp_difference/mean": 0.012091459706425667, "step": 482 }, { "clip_ratio/high_max": 0.010245901241432875, "clip_ratio/high_mean": 0.010245901241432875, "clip_ratio/low_mean": 0.0014594576205126941, "clip_ratio/low_min": 0.0014594576205126941, "clip_ratio/region_mean": 0.01170535886194557, "completions/clipped_ratio": 0.0, "completions/max_length": 183.0, "completions/max_terminated_length": 183.0, "completions/mean_length": 180.25, "completions/mean_terminated_length": 180.25, "completions/min_length": 161.0, "completions/min_terminated_length": 161.0, "entropy": 0.0649343291297555, "epoch": 0.018651529193697867, "frac_reward_zero_std": 0.0, "grad_norm": 3.103454828262329, "learning_rate": 8.539393939393939e-06, "loss": 0.0021, "num_tokens": 1043145.0, "reward": 0.9860587120056152, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9860587120056152, "reward_meter_std": 0.00629128934815526, "reward_std": 0.006291288882493973, "reward_total_composite_mean": 0.9860587120056152, "reward_total_composite_std": 0.00629128934815526, "reward_total_mean": 0.9860587120056152, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9860587120056152, "rewards/meter/std": 0.00629128934815526, "rewards/total_composite/mean": 0.9860587120056152, "rewards/total_composite/std": 0.00629128934815526, "sampling/importance_sampling_ratio/max": 1.7320133447647095, "sampling/importance_sampling_ratio/mean": 1.0001847743988037, "sampling/importance_sampling_ratio/min": 0.41738274693489075, "sampling/sampling_logp_difference/max": 0.8737516403198242, "sampling/sampling_logp_difference/mean": 0.011846227571368217, "step": 483 }, { "clip_ratio/high_max": 0.059974749106913805, "clip_ratio/high_mean": 0.059974749106913805, "clip_ratio/low_mean": 0.029679158004000783, "clip_ratio/low_min": 0.029679158004000783, "clip_ratio/region_mean": 0.08965390711091459, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.9045082405209541, "epoch": 0.01869014519616929, "frac_reward_zero_std": 0.0, "grad_norm": 9.362083435058594, "learning_rate": 8.536363636363636e-06, "loss": 0.0128, "num_tokens": 1044895.0, "reward": 0.6785883903503418, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7406154870986938, "reward_meter_std": 0.36668860912323, "reward_std": 0.35991325974464417, "reward_total_composite_mean": 0.6785883903503418, "reward_total_composite_std": 0.3599132299423218, "reward_total_mean": 0.6785883903503418, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7406154870986938, "rewards/meter/std": 0.36668860912323, "rewards/total_composite/mean": 0.6785883903503418, "rewards/total_composite/std": 0.3599132299423218, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0062280893325806, "sampling/importance_sampling_ratio/min": 0.25423842668533325, "sampling/sampling_logp_difference/max": 1.3694827556610107, "sampling/sampling_logp_difference/mean": 0.10126848518848419, "step": 484 }, { "clip_ratio/high_max": 0.07197061297483742, "clip_ratio/high_mean": 0.07197061297483742, "clip_ratio/low_mean": 0.015625, "clip_ratio/low_min": 0.015625, "clip_ratio/region_mean": 0.08759561297483742, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 61.375, "completions/mean_terminated_length": 61.375, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.5880453810095787, "epoch": 0.018728761198640715, "frac_reward_zero_std": 0.0, "grad_norm": 11.369089126586914, "learning_rate": 8.533333333333335e-06, "loss": -0.0278, "num_tokens": 1046762.0, "reward": 0.8861362338066101, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8861362338066101, "reward_meter_std": 0.2756871283054352, "reward_std": 0.2756870985031128, "reward_total_composite_mean": 0.8861362338066101, "reward_total_composite_std": 0.2756871283054352, "reward_total_mean": 0.8861362338066101, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8861362338066101, "rewards/meter/std": 0.2756871283054352, "rewards/total_composite/mean": 0.8861362338066101, "rewards/total_composite/std": 0.2756871283054352, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9978853464126587, "sampling/importance_sampling_ratio/min": 0.26445987820625305, "sampling/sampling_logp_difference/max": 1.3300657272338867, "sampling/sampling_logp_difference/mean": 0.08699135482311249, "step": 485 }, { "clip_ratio/high_max": 0.030109542072750628, "clip_ratio/high_mean": 0.030109542072750628, "clip_ratio/low_mean": 0.02085444121621549, "clip_ratio/low_min": 0.02085444121621549, "clip_ratio/region_mean": 0.05096398328896612, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 99.5, "completions/mean_terminated_length": 99.5, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.2146557718515396, "epoch": 0.018767377201112143, "frac_reward_zero_std": 0.0, "grad_norm": 4.333034992218018, "learning_rate": 8.53030303030303e-06, "loss": -0.026, "num_tokens": 1048822.0, "reward": 0.7474471926689148, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.786818265914917, "reward_meter_std": 0.3528897166252136, "reward_std": 0.3502931594848633, "reward_total_composite_mean": 0.7474471926689148, "reward_total_composite_std": 0.35029318928718567, "reward_total_mean": 0.7474471926689148, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.786818265914917, "rewards/meter/std": 0.3528897166252136, "rewards/total_composite/mean": 0.7474471926689148, "rewards/total_composite/std": 0.35029318928718567, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002170443534851, "sampling/importance_sampling_ratio/min": 0.22690196335315704, "sampling/sampling_logp_difference/max": 1.4832372665405273, "sampling/sampling_logp_difference/mean": 0.0431230328977108, "step": 486 }, { "clip_ratio/high_max": 0.013687330298125744, "clip_ratio/high_mean": 0.013687330298125744, "clip_ratio/low_mean": 0.002577319508418441, "clip_ratio/low_min": 0.002577319508418441, "clip_ratio/region_mean": 0.016264649806544185, "completions/clipped_ratio": 0.0, "completions/max_length": 439.0, "completions/max_terminated_length": 439.0, "completions/mean_length": 400.125, "completions/mean_terminated_length": 400.125, "completions/min_length": 344.0, "completions/min_terminated_length": 344.0, "entropy": 0.12399658421054482, "epoch": 0.018805993203583567, "frac_reward_zero_std": 0.0, "grad_norm": 1.3781453371047974, "learning_rate": 8.527272727272728e-06, "loss": -0.0076, "num_tokens": 1053671.0, "reward": 0.6021700501441956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9634721279144287, "reward_meter_std": 0.0763736218214035, "reward_std": 0.047733522951602936, "reward_total_composite_mean": 0.6021700501441956, "reward_total_composite_std": 0.047733522951602936, "reward_total_mean": 0.6021700501441956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9634721279144287, "rewards/meter/std": 0.0763736218214035, "rewards/total_composite/mean": 0.6021700501441956, "rewards/total_composite/std": 0.047733522951602936, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0029096603393555, "sampling/importance_sampling_ratio/min": 0.25609996914863586, "sampling/sampling_logp_difference/max": 1.362187385559082, "sampling/sampling_logp_difference/mean": 0.018493853509426117, "step": 487 }, { "clip_ratio/high_max": 0.08754385984502733, "clip_ratio/high_mean": 0.08754385984502733, "clip_ratio/low_mean": 0.019999999552965164, "clip_ratio/low_min": 0.019999999552965164, "clip_ratio/region_mean": 0.10754385939799249, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 56.375, "completions/mean_terminated_length": 56.375, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.9573509357869625, "epoch": 0.01884460920605499, "frac_reward_zero_std": 0.0, "grad_norm": 12.93553352355957, "learning_rate": 8.524242424242425e-06, "loss": -0.0275, "num_tokens": 1055530.0, "reward": 0.9578395485877991, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9578395485877991, "reward_meter_std": 0.06491218507289886, "reward_std": 0.06491218507289886, "reward_total_composite_mean": 0.9578395485877991, "reward_total_composite_std": 0.06491218507289886, "reward_total_mean": 0.9578395485877991, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9578395485877991, "rewards/meter/std": 0.06491218507289886, "rewards/total_composite/mean": 0.9578395485877991, "rewards/total_composite/std": 0.06491218507289886, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005127191543579, "sampling/importance_sampling_ratio/min": 0.29221248626708984, "sampling/sampling_logp_difference/max": 1.230273962020874, "sampling/sampling_logp_difference/mean": 0.11469703167676926, "step": 488 }, { "clip_ratio/high_max": 0.02726867760065943, "clip_ratio/high_mean": 0.02726867760065943, "clip_ratio/low_mean": 0.00903480825945735, "clip_ratio/low_min": 0.00903480825945735, "clip_ratio/region_mean": 0.03630348586011678, "completions/clipped_ratio": 0.0, "completions/max_length": 150.0, "completions/max_terminated_length": 150.0, "completions/mean_length": 133.875, "completions/mean_terminated_length": 133.875, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.25978855788707733, "epoch": 0.018883225208526415, "frac_reward_zero_std": 0.0, "grad_norm": 3.998319625854492, "learning_rate": 8.521212121212123e-06, "loss": 0.0046, "num_tokens": 1058017.0, "reward": 0.9484606981277466, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9484606981277466, "reward_meter_std": 0.09933194518089294, "reward_std": 0.09933193773031235, "reward_total_composite_mean": 0.9484606981277466, "reward_total_composite_std": 0.09933194518089294, "reward_total_mean": 0.9484606981277466, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9484606981277466, "rewards/meter/std": 0.09933194518089294, "rewards/total_composite/mean": 0.9484606981277466, "rewards/total_composite/std": 0.09933194518089294, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006198763847351, "sampling/importance_sampling_ratio/min": 0.23934954404830933, "sampling/sampling_logp_difference/max": 1.4298303127288818, "sampling/sampling_logp_difference/mean": 0.0375179685652256, "step": 489 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0007530120201408863, "clip_ratio/low_min": 0.0007530120201408863, "clip_ratio/region_mean": 0.0007530120201408863, "completions/clipped_ratio": 0.875, "completions/max_length": 512.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 510.25, "completions/mean_terminated_length": 498.0, "completions/min_length": 498.0, "completions/min_terminated_length": 498.0, "entropy": 0.00822315365076065, "epoch": 0.01892184121099784, "frac_reward_zero_std": 0.0, "grad_norm": 0.38429051637649536, "learning_rate": 8.518181818181818e-06, "loss": 0.1271, "num_tokens": 1060315.0, "reward": 0.8464405536651611, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8557692170143127, "reward_count_adherence_std": 0.08661473542451859, "reward_meter_mean": 0.9891150593757629, "reward_meter_std": 0.00664390018209815, "reward_std": 0.08564455062150955, "reward_total_composite_mean": 0.8464405536651611, "reward_total_composite_std": 0.08564455062150955, "reward_total_mean": 0.8464405536651611, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8557692170143127, "rewards/count_adherence/std": 0.08661473542451859, "rewards/meter/mean": 0.9891150593757629, "rewards/meter/std": 0.00664390018209815, "rewards/total_composite/mean": 0.8464405536651611, "rewards/total_composite/std": 0.08564455062150955, "sampling/importance_sampling_ratio/max": 1.7250312566757202, "sampling/importance_sampling_ratio/mean": 0.9997017979621887, "sampling/importance_sampling_ratio/min": 0.3699522316455841, "sampling/sampling_logp_difference/max": 0.9943814277648926, "sampling/sampling_logp_difference/mean": 0.011192544363439083, "step": 490 }, { "clip_ratio/high_max": 0.03605138626880944, "clip_ratio/high_mean": 0.03605138626880944, "clip_ratio/low_mean": 0.043553632916882634, "clip_ratio/low_min": 0.043553632916882634, "clip_ratio/region_mean": 0.07960501918569207, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 93.25, "completions/mean_terminated_length": 93.25, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.900152787566185, "epoch": 0.018960457213469263, "frac_reward_zero_std": 0.0, "grad_norm": 4.761613368988037, "learning_rate": 8.515151515151517e-06, "loss": 0.0351, "num_tokens": 1062421.0, "reward": 0.6782705783843994, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6782705783843994, "reward_meter_std": 0.363802045583725, "reward_std": 0.363802045583725, "reward_total_composite_mean": 0.6782705783843994, "reward_total_composite_std": 0.363802045583725, "reward_total_mean": 0.6782705783843994, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6782705783843994, "rewards/meter/std": 0.363802045583725, "rewards/total_composite/mean": 0.6782705783843994, "rewards/total_composite/std": 0.363802045583725, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0159639120101929, "sampling/importance_sampling_ratio/min": 0.26024797558784485, "sampling/sampling_logp_difference/max": 1.3461203575134277, "sampling/sampling_logp_difference/mean": 0.08261455595493317, "step": 491 }, { "clip_ratio/high_max": 0.03652191942092031, "clip_ratio/high_mean": 0.03652191942092031, "clip_ratio/low_mean": 0.032360111363232136, "clip_ratio/low_min": 0.032360111363232136, "clip_ratio/region_mean": 0.06888203078415245, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 68.5, "completions/mean_terminated_length": 68.5, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.7679142504930496, "epoch": 0.018999073215940687, "frac_reward_zero_std": 0.0, "grad_norm": 10.827360153198242, "learning_rate": 8.512121212121213e-06, "loss": 0.0363, "num_tokens": 1064305.0, "reward": 0.9250147342681885, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9250147342681885, "reward_meter_std": 0.151312917470932, "reward_std": 0.1513129323720932, "reward_total_composite_mean": 0.9250147342681885, "reward_total_composite_std": 0.151312917470932, "reward_total_mean": 0.9250147342681885, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9250147342681885, "rewards/meter/std": 0.151312917470932, "rewards/total_composite/mean": 0.9250147342681885, "rewards/total_composite/std": 0.151312917470932, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.018401861190796, "sampling/importance_sampling_ratio/min": 0.1545751392841339, "sampling/sampling_logp_difference/max": 1.867074966430664, "sampling/sampling_logp_difference/mean": 0.10603255778551102, "step": 492 }, { "clip_ratio/high_max": 0.039439259795472026, "clip_ratio/high_mean": 0.039439259795472026, "clip_ratio/low_mean": 0.016476215794682503, "clip_ratio/low_min": 0.016476215794682503, "clip_ratio/region_mean": 0.05591547559015453, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 68.875, "completions/mean_terminated_length": 68.875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.5212747771292925, "epoch": 0.01903768921841211, "frac_reward_zero_std": 0.0, "grad_norm": 5.6415114402771, "learning_rate": 8.50909090909091e-06, "loss": -0.0015, "num_tokens": 1066128.0, "reward": 0.9896060228347778, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9896060228347778, "reward_meter_std": 0.009286360815167427, "reward_std": 0.009286360815167427, "reward_total_composite_mean": 0.9896060228347778, "reward_total_composite_std": 0.009286360815167427, "reward_total_mean": 0.9896060228347778, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9896060228347778, "rewards/meter/std": 0.009286360815167427, "rewards/total_composite/mean": 0.9896060228347778, "rewards/total_composite/std": 0.009286360815167427, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0087600946426392, "sampling/importance_sampling_ratio/min": 0.14200963079929352, "sampling/sampling_logp_difference/max": 1.9518604278564453, "sampling/sampling_logp_difference/mean": 0.07539483904838562, "step": 493 }, { "clip_ratio/high_max": 0.034729897044599056, "clip_ratio/high_mean": 0.034729897044599056, "clip_ratio/low_mean": 0.03369492199271917, "clip_ratio/low_min": 0.03369492199271917, "clip_ratio/region_mean": 0.06842481903731823, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 58.75, "completions/mean_terminated_length": 58.75, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.47727346792817116, "epoch": 0.019076305220883535, "frac_reward_zero_std": 0.0, "grad_norm": 6.613814353942871, "learning_rate": 8.506060606060607e-06, "loss": 0.0225, "num_tokens": 1067822.0, "reward": 0.9344394207000732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9344394207000732, "reward_meter_std": 0.060838740319013596, "reward_std": 0.060838740319013596, "reward_total_composite_mean": 0.9344394207000732, "reward_total_composite_std": 0.060838740319013596, "reward_total_mean": 0.9344394207000732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9344394207000732, "rewards/meter/std": 0.060838740319013596, "rewards/total_composite/mean": 0.9344394207000732, "rewards/total_composite/std": 0.060838740319013596, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.014532446861267, "sampling/importance_sampling_ratio/min": 0.14271166920661926, "sampling/sampling_logp_difference/max": 1.9469289779663086, "sampling/sampling_logp_difference/mean": 0.06783849745988846, "step": 494 }, { "clip_ratio/high_max": 0.027538669761270285, "clip_ratio/high_mean": 0.027538669761270285, "clip_ratio/low_mean": 0.024894393514841795, "clip_ratio/low_min": 0.024894393514841795, "clip_ratio/region_mean": 0.05243306327611208, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 30.875, "completions/mean_terminated_length": 30.875, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.3115545194596052, "epoch": 0.01911492122335496, "frac_reward_zero_std": 0.0, "grad_norm": 12.706208229064941, "learning_rate": 8.503030303030304e-06, "loss": -0.0075, "num_tokens": 1069485.0, "reward": 0.9881634712219238, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9881634712219238, "reward_meter_std": 0.007193571422249079, "reward_std": 0.007193575147539377, "reward_total_composite_mean": 0.9881634712219238, "reward_total_composite_std": 0.007193571422249079, "reward_total_mean": 0.9881634712219238, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9881634712219238, "rewards/meter/std": 0.007193571422249079, "rewards/total_composite/mean": 0.9881634712219238, "rewards/total_composite/std": 0.007193571422249079, "sampling/importance_sampling_ratio/max": 1.6685515642166138, "sampling/importance_sampling_ratio/mean": 1.0025434494018555, "sampling/importance_sampling_ratio/min": 0.25602710247039795, "sampling/sampling_logp_difference/max": 1.3624719381332397, "sampling/sampling_logp_difference/mean": 0.060055576264858246, "step": 495 }, { "clip_ratio/high_max": 0.013113626977428794, "clip_ratio/high_mean": 0.013113626977428794, "clip_ratio/low_mean": 0.01094052626285702, "clip_ratio/low_min": 0.01094052626285702, "clip_ratio/region_mean": 0.024054153240285814, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 92.5, "completions/mean_terminated_length": 92.5, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.17256110534071922, "epoch": 0.019153537225826384, "frac_reward_zero_std": 0.0, "grad_norm": 3.555715322494507, "learning_rate": 8.5e-06, "loss": -0.0266, "num_tokens": 1071673.0, "reward": 0.9900758862495422, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9900758862495422, "reward_meter_std": 0.004209411796182394, "reward_std": 0.004209410399198532, "reward_total_composite_mean": 0.9900758862495422, "reward_total_composite_std": 0.004209411796182394, "reward_total_mean": 0.9900758862495422, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9900758862495422, "rewards/meter/std": 0.004209411796182394, "rewards/total_composite/mean": 0.9900758862495422, "rewards/total_composite/std": 0.004209411796182394, "sampling/importance_sampling_ratio/max": 1.8225340843200684, "sampling/importance_sampling_ratio/mean": 1.0048907995224, "sampling/importance_sampling_ratio/min": 0.42399612069129944, "sampling/sampling_logp_difference/max": 0.8580310344696045, "sampling/sampling_logp_difference/mean": 0.02267627976834774, "step": 496 }, { "clip_ratio/high_max": 0.055402441415935755, "clip_ratio/high_mean": 0.055402441415935755, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/region_mean": 0.06254529859870672, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 73.25, "completions/mean_terminated_length": 73.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.5875852480530739, "epoch": 0.019192153228297808, "frac_reward_zero_std": 0.0, "grad_norm": 4.990721225738525, "learning_rate": 8.496969696969697e-06, "loss": -0.0116, "num_tokens": 1073467.0, "reward": 0.9253207445144653, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9253207445144653, "reward_meter_std": 0.19340236485004425, "reward_std": 0.19340236485004425, "reward_total_composite_mean": 0.9253207445144653, "reward_total_composite_std": 0.19340236485004425, "reward_total_mean": 0.9253207445144653, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9253207445144653, "rewards/meter/std": 0.19340236485004425, "rewards/total_composite/mean": 0.9253207445144653, "rewards/total_composite/std": 0.19340236485004425, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0199859142303467, "sampling/importance_sampling_ratio/min": 0.06627547740936279, "sampling/sampling_logp_difference/max": 2.713935375213623, "sampling/sampling_logp_difference/mean": 0.08409885317087173, "step": 497 }, { "clip_ratio/high_max": 0.010509460349567235, "clip_ratio/high_mean": 0.010509460349567235, "clip_ratio/low_mean": 0.01110077218618244, "clip_ratio/low_min": 0.01110077218618244, "clip_ratio/region_mean": 0.021610232535749674, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 91.125, "completions/mean_terminated_length": 91.125, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.1501608146354556, "epoch": 0.019230769230769232, "frac_reward_zero_std": 0.0, "grad_norm": 3.0449001789093018, "learning_rate": 8.493939393939394e-06, "loss": -0.0224, "num_tokens": 1075516.0, "reward": 0.9902229905128479, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9902229905128479, "reward_meter_std": 0.003602617187425494, "reward_std": 0.0036026162561029196, "reward_total_composite_mean": 0.9902229905128479, "reward_total_composite_std": 0.003602617187425494, "reward_total_mean": 0.9902229905128479, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9902229905128479, "rewards/meter/std": 0.003602617187425494, "rewards/total_composite/mean": 0.9902229905128479, "rewards/total_composite/std": 0.003602617187425494, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.997484564781189, "sampling/importance_sampling_ratio/min": 0.2278515249490738, "sampling/sampling_logp_difference/max": 1.4790611267089844, "sampling/sampling_logp_difference/mean": 0.023890873417258263, "step": 498 }, { "clip_ratio/high_max": 0.01690437039360404, "clip_ratio/high_mean": 0.01690437039360404, "clip_ratio/low_mean": 0.005158253246918321, "clip_ratio/low_min": 0.005158253246918321, "clip_ratio/region_mean": 0.02206262364052236, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 68.375, "completions/mean_terminated_length": 68.375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.23555130790919065, "epoch": 0.019269385233240656, "frac_reward_zero_std": 0.0, "grad_norm": 4.565327167510986, "learning_rate": 8.490909090909092e-06, "loss": 0.0379, "num_tokens": 1077319.0, "reward": 0.9964392185211182, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9964392185211182, "reward_meter_std": 0.0012064595939591527, "reward_std": 0.0012064601760357618, "reward_total_composite_mean": 0.9964392185211182, "reward_total_composite_std": 0.0012064595939591527, "reward_total_mean": 0.9964392185211182, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9964392185211182, "rewards/meter/std": 0.0012064595939591527, "rewards/total_composite/mean": 0.9964392185211182, "rewards/total_composite/std": 0.0012064595939591527, "sampling/importance_sampling_ratio/max": 1.8629289865493774, "sampling/importance_sampling_ratio/mean": 1.0085853338241577, "sampling/importance_sampling_ratio/min": 0.21121247112751007, "sampling/sampling_logp_difference/max": 1.5548906326293945, "sampling/sampling_logp_difference/mean": 0.032802514731884, "step": 499 }, { "clip_ratio/high_max": 0.03529414697550237, "clip_ratio/high_mean": 0.03529414697550237, "clip_ratio/low_mean": 0.01627026009373367, "clip_ratio/low_min": 0.01627026009373367, "clip_ratio/region_mean": 0.05156440706923604, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 77.5, "completions/mean_terminated_length": 77.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.35560205206274986, "epoch": 0.01930800123571208, "frac_reward_zero_std": 0.0, "grad_norm": 4.640279769897461, "learning_rate": 8.487878787878789e-06, "loss": -0.0131, "num_tokens": 1079123.0, "reward": 0.9947643280029297, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9947643280029297, "reward_meter_std": 0.0026077541988343, "reward_std": 0.0026077530346810818, "reward_total_composite_mean": 0.9947643280029297, "reward_total_composite_std": 0.0026077541988343, "reward_total_mean": 0.9947643280029297, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9947643280029297, "rewards/meter/std": 0.0026077541988343, "rewards/total_composite/mean": 0.9947643280029297, "rewards/total_composite/std": 0.0026077541988343, "sampling/importance_sampling_ratio/max": 1.9909405708312988, "sampling/importance_sampling_ratio/mean": 1.0034871101379395, "sampling/importance_sampling_ratio/min": 0.24968643486499786, "sampling/sampling_logp_difference/max": 1.3875494003295898, "sampling/sampling_logp_difference/mean": 0.05810660123825073, "step": 500 }, { "epoch": 0.01930800123571208, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.27884615384615385, "eval_completions/max_length": 504.3076923076923, "eval_completions/max_terminated_length": 360.61538461538464, "eval_completions/mean_length": 282.3942307692308, "eval_completions/mean_terminated_length": 189.76557100736179, "eval_completions/min_length": 56.53846153846154, "eval_completions/min_terminated_length": 56.53846153846154, "eval_entropy": 0.2863846653356002, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 1079123.0, "eval_reward": 0.5957829631291903, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.06280488005051246, "eval_reward_count_adherence_mean": 0.7540726845081036, "eval_reward_count_adherence_std": 0.26829315836612994, "eval_reward_meter_mean": 0.806977744285877, "eval_reward_meter_std": 0.28674349504021496, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5957829631291903, "eval_reward_total_composite_std": 0.3549887262857877, "eval_reward_total_mean": 0.5957829631291903, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.06280488005051246, "eval_rewards/count_adherence/mean": 0.7540726845081036, "eval_rewards/count_adherence/std": 0.26829315836612994, "eval_rewards/meter/mean": 0.806977744285877, "eval_rewards/meter/std": 0.28674349504021496, "eval_rewards/total_composite/mean": 0.5957829631291903, "eval_rewards/total_composite/std": 0.3549887262857877, "eval_runtime": 93.945, "eval_samples_per_second": 1.107, "eval_sampling/importance_sampling_ratio/max": 1.4029179719778209, "eval_sampling/importance_sampling_ratio/mean": 1.006858468055725, "eval_sampling/importance_sampling_ratio/min": 0.38570847190343416, "eval_sampling/sampling_logp_difference/max": 1.013349202963022, "eval_sampling/sampling_logp_difference/mean": 0.02426047422564947, "eval_steps_per_second": 0.138, "step": 500 }, { "clip_ratio/high_max": 0.023434267612174153, "clip_ratio/high_mean": 0.023434267612174153, "clip_ratio/low_mean": 0.006404084269888699, "clip_ratio/low_min": 0.006404084269888699, "clip_ratio/region_mean": 0.029838351882062852, "completions/clipped_ratio": 0.0, "completions/max_length": 119.0, "completions/max_terminated_length": 119.0, "completions/mean_length": 106.125, "completions/mean_terminated_length": 106.125, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.22485548444092274, "epoch": 0.019346617238183504, "frac_reward_zero_std": 0.0, "grad_norm": 5.740342617034912, "learning_rate": 8.484848484848486e-06, "loss": 0.0695, "num_tokens": 1081252.0, "reward": 0.78095543384552, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.78095543384552, "reward_meter_std": 0.2571522295475006, "reward_std": 0.2571522295475006, "reward_total_composite_mean": 0.78095543384552, "reward_total_composite_std": 0.2571522295475006, "reward_total_mean": 0.78095543384552, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.78095543384552, "rewards/meter/std": 0.2571522295475006, "rewards/total_composite/mean": 0.78095543384552, "rewards/total_composite/std": 0.2571522295475006, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000687599182129, "sampling/importance_sampling_ratio/min": 0.14426499605178833, "sampling/sampling_logp_difference/max": 1.936103343963623, "sampling/sampling_logp_difference/mean": 0.0388408899307251, "step": 501 }, { "clip_ratio/high_max": 0.04436044883914292, "clip_ratio/high_mean": 0.04436044883914292, "clip_ratio/low_mean": 0.02120496891438961, "clip_ratio/low_min": 0.02120496891438961, "clip_ratio/region_mean": 0.06556541775353253, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 74.125, "completions/mean_terminated_length": 74.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.6134384572505951, "epoch": 0.019385233240654928, "frac_reward_zero_std": 0.0, "grad_norm": 5.4612603187561035, "learning_rate": 8.481818181818182e-06, "loss": -0.0119, "num_tokens": 1083101.0, "reward": 0.9958293437957764, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958293437957764, "reward_meter_std": 0.0022998028434813023, "reward_std": 0.0022997905034571886, "reward_total_composite_mean": 0.9958293437957764, "reward_total_composite_std": 0.0022998028434813023, "reward_total_mean": 0.9958293437957764, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958293437957764, "rewards/meter/std": 0.0022998028434813023, "rewards/total_composite/mean": 0.9958293437957764, "rewards/total_composite/std": 0.0022998028434813023, "sampling/importance_sampling_ratio/max": 1.5113362073898315, "sampling/importance_sampling_ratio/mean": 0.9978984594345093, "sampling/importance_sampling_ratio/min": 0.2680939733982086, "sampling/sampling_logp_difference/max": 1.3164176940917969, "sampling/sampling_logp_difference/mean": 0.07660207152366638, "step": 502 }, { "clip_ratio/high_max": 0.016571022104471922, "clip_ratio/high_mean": 0.016571022104471922, "clip_ratio/low_mean": 0.011187375290319324, "clip_ratio/low_min": 0.011187375290319324, "clip_ratio/region_mean": 0.027758397394791245, "completions/clipped_ratio": 0.0, "completions/max_length": 243.0, "completions/max_terminated_length": 243.0, "completions/mean_length": 224.0, "completions/mean_terminated_length": 224.0, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "entropy": 0.2083021691069007, "epoch": 0.019423849243126352, "frac_reward_zero_std": 0.0, "grad_norm": 2.5192019939422607, "learning_rate": 8.478787878787879e-06, "loss": 0.0356, "num_tokens": 1086421.0, "reward": 0.7093769311904907, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7916666269302368, "reward_count_adherence_std": 0.07715165615081787, "reward_meter_mean": 0.9007662534713745, "reward_meter_std": 0.15409159660339355, "reward_std": 0.12365645170211792, "reward_total_composite_mean": 0.7093769311904907, "reward_total_composite_std": 0.12365645170211792, "reward_total_mean": 0.7093769311904907, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7916666269302368, "rewards/count_adherence/std": 0.07715165615081787, "rewards/meter/mean": 0.9007662534713745, "rewards/meter/std": 0.15409159660339355, "rewards/total_composite/mean": 0.7093769311904907, "rewards/total_composite/std": 0.12365645170211792, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005336880683899, "sampling/importance_sampling_ratio/min": 0.49309608340263367, "sampling/sampling_logp_difference/max": 1.0005009174346924, "sampling/sampling_logp_difference/mean": 0.02353961206972599, "step": 503 }, { "clip_ratio/high_max": 0.04974571894854307, "clip_ratio/high_mean": 0.04974571894854307, "clip_ratio/low_mean": 0.0136233662487939, "clip_ratio/low_min": 0.0136233662487939, "clip_ratio/region_mean": 0.06336908519733697, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 77.625, "completions/mean_terminated_length": 77.625, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.6513996087014675, "epoch": 0.019462465245597776, "frac_reward_zero_std": 0.0, "grad_norm": 7.049271583557129, "learning_rate": 8.475757575757576e-06, "loss": 0.0237, "num_tokens": 1088370.0, "reward": 0.7662866115570068, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7662866115570068, "reward_meter_std": 0.42396625876426697, "reward_std": 0.4239662289619446, "reward_total_composite_mean": 0.7662866115570068, "reward_total_composite_std": 0.42396625876426697, "reward_total_mean": 0.7662866115570068, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7662866115570068, "rewards/meter/std": 0.42396625876426697, "rewards/total_composite/mean": 0.7662866115570068, "rewards/total_composite/std": 0.42396625876426697, "sampling/importance_sampling_ratio/max": 1.7661075592041016, "sampling/importance_sampling_ratio/mean": 1.0156745910644531, "sampling/importance_sampling_ratio/min": 0.2086387276649475, "sampling/sampling_logp_difference/max": 1.5671510696411133, "sampling/sampling_logp_difference/mean": 0.07983585447072983, "step": 504 }, { "clip_ratio/high_max": 0.020820592530071735, "clip_ratio/high_mean": 0.020820592530071735, "clip_ratio/low_mean": 0.0246689785271883, "clip_ratio/low_min": 0.0246689785271883, "clip_ratio/region_mean": 0.045489571057260036, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 55.625, "completions/mean_terminated_length": 55.625, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.5484390016645193, "epoch": 0.0195010812480692, "frac_reward_zero_std": 0.0, "grad_norm": 8.706897735595703, "learning_rate": 8.472727272727274e-06, "loss": 0.0374, "num_tokens": 1090079.0, "reward": 0.9907145500183105, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9907145500183105, "reward_meter_std": 0.0025706093292683363, "reward_std": 0.002570599550381303, "reward_total_composite_mean": 0.9907145500183105, "reward_total_composite_std": 0.0025706093292683363, "reward_total_mean": 0.9907145500183105, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9907145500183105, "rewards/meter/std": 0.0025706093292683363, "rewards/total_composite/mean": 0.9907145500183105, "rewards/total_composite/std": 0.0025706093292683363, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0099107027053833, "sampling/importance_sampling_ratio/min": 0.22027796506881714, "sampling/sampling_logp_difference/max": 1.5128650665283203, "sampling/sampling_logp_difference/mean": 0.06765672564506531, "step": 505 }, { "clip_ratio/high_max": 0.04047576058655977, "clip_ratio/high_mean": 0.04047576058655977, "clip_ratio/low_mean": 0.04866071417927742, "clip_ratio/low_min": 0.04866071417927742, "clip_ratio/region_mean": 0.08913647476583719, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 54.25, "completions/mean_terminated_length": 54.25, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.9682378023862839, "epoch": 0.019539697250540625, "frac_reward_zero_std": 0.0, "grad_norm": 9.833094596862793, "learning_rate": 8.46969696969697e-06, "loss": -0.1105, "num_tokens": 1091745.0, "reward": 0.8061291575431824, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8061291575431824, "reward_meter_std": 0.3550299108028412, "reward_std": 0.3550299406051636, "reward_total_composite_mean": 0.8061291575431824, "reward_total_composite_std": 0.3550299108028412, "reward_total_mean": 0.8061291575431824, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8061291575431824, "rewards/meter/std": 0.3550299108028412, "rewards/total_composite/mean": 0.8061291575431824, "rewards/total_composite/std": 0.3550299108028412, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0167040824890137, "sampling/importance_sampling_ratio/min": 0.2069854736328125, "sampling/sampling_logp_difference/max": 1.5751066207885742, "sampling/sampling_logp_difference/mean": 0.08280141651630402, "step": 506 }, { "clip_ratio/high_max": 0.06833233823999763, "clip_ratio/high_mean": 0.06833233823999763, "clip_ratio/low_mean": 0.019542983267456293, "clip_ratio/low_min": 0.019542983267456293, "clip_ratio/region_mean": 0.08787532150745392, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.7916622683405876, "epoch": 0.01957831325301205, "frac_reward_zero_std": 0.0, "grad_norm": 7.7388482093811035, "learning_rate": 8.466666666666668e-06, "loss": 0.0027, "num_tokens": 1093764.0, "reward": 0.9402273893356323, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9402273893356323, "reward_meter_std": 0.08104795962572098, "reward_std": 0.08104795962572098, "reward_total_composite_mean": 0.9402273893356323, "reward_total_composite_std": 0.08104795962572098, "reward_total_mean": 0.9402273893356323, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9402273893356323, "rewards/meter/std": 0.08104795962572098, "rewards/total_composite/mean": 0.9402273893356323, "rewards/total_composite/std": 0.08104795962572098, "sampling/importance_sampling_ratio/max": 1.8226356506347656, "sampling/importance_sampling_ratio/mean": 1.0046333074569702, "sampling/importance_sampling_ratio/min": 0.24624574184417725, "sampling/sampling_logp_difference/max": 1.4014253616333008, "sampling/sampling_logp_difference/mean": 0.09213840216398239, "step": 507 }, { "clip_ratio/high_max": 0.02592427283525467, "clip_ratio/high_mean": 0.02592427283525467, "clip_ratio/low_mean": 0.015961318742483854, "clip_ratio/low_min": 0.015961318742483854, "clip_ratio/region_mean": 0.041885591577738523, "completions/clipped_ratio": 0.0, "completions/max_length": 193.0, "completions/max_terminated_length": 193.0, "completions/mean_length": 156.0, "completions/mean_terminated_length": 156.0, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "entropy": 0.37319300696253777, "epoch": 0.019616929255483473, "frac_reward_zero_std": 0.0, "grad_norm": 3.3655362129211426, "learning_rate": 8.463636363636364e-06, "loss": -0.0493, "num_tokens": 1096436.0, "reward": 0.9363090991973877, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9674587845802307, "reward_meter_std": 0.08305719494819641, "reward_std": 0.11212778836488724, "reward_total_composite_mean": 0.9363090991973877, "reward_total_composite_std": 0.11212779581546783, "reward_total_mean": 0.9363090991973877, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9674587845802307, "rewards/meter/std": 0.08305719494819641, "rewards/total_composite/mean": 0.9363090991973877, "rewards/total_composite/std": 0.11212779581546783, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0092945098876953, "sampling/importance_sampling_ratio/min": 0.3582862913608551, "sampling/sampling_logp_difference/max": 1.0351219177246094, "sampling/sampling_logp_difference/mean": 0.04139424115419388, "step": 508 }, { "clip_ratio/high_max": 0.09690898563712835, "clip_ratio/high_mean": 0.09690898563712835, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.09690898563712835, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 121.625, "completions/mean_terminated_length": 65.85714721679688, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 1.1623116582632065, "epoch": 0.019655545257954897, "frac_reward_zero_std": 0.0, "grad_norm": 1.5039535760879517, "learning_rate": 8.460606060606061e-06, "loss": -0.1651, "num_tokens": 1098089.0, "reward": 0.850326418876648, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8553266525268555, "reward_meter_std": 0.3322550356388092, "reward_std": 0.3462829291820526, "reward_total_composite_mean": 0.850326418876648, "reward_total_composite_std": 0.3462829291820526, "reward_total_mean": 0.850326418876648, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8553266525268555, "rewards/meter/std": 0.3322550356388092, "rewards/total_composite/mean": 0.850326418876648, "rewards/total_composite/std": 0.3462829291820526, "sampling/importance_sampling_ratio/max": 1.746443271636963, "sampling/importance_sampling_ratio/mean": 1.0222753286361694, "sampling/importance_sampling_ratio/min": 0.27474337816238403, "sampling/sampling_logp_difference/max": 1.2919178009033203, "sampling/sampling_logp_difference/mean": 0.12041088938713074, "step": 509 }, { "clip_ratio/high_max": 0.006500857998616993, "clip_ratio/high_mean": 0.006500857998616993, "clip_ratio/low_mean": 0.00692010746570304, "clip_ratio/low_min": 0.00692010746570304, "clip_ratio/region_mean": 0.013420965464320034, "completions/clipped_ratio": 0.0, "completions/max_length": 505.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 447.125, "completions/mean_terminated_length": 447.125, "completions/min_length": 398.0, "completions/min_terminated_length": 398.0, "entropy": 0.1255875793285668, "epoch": 0.01969416126042632, "frac_reward_zero_std": 0.0, "grad_norm": 1.3078835010528564, "learning_rate": 8.457575757575758e-06, "loss": 0.0177, "num_tokens": 1103402.0, "reward": 0.3376147150993347, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.4464285969734192, "reward_count_adherence_std": 0.11921756714582443, "reward_meter_mean": 0.7473684549331665, "reward_meter_std": 0.4575249254703522, "reward_std": 0.22579078376293182, "reward_total_composite_mean": 0.3376147150993347, "reward_total_composite_std": 0.22579079866409302, "reward_total_mean": 0.3376147150993347, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.4464285969734192, "rewards/count_adherence/std": 0.11921756714582443, "rewards/meter/mean": 0.7473684549331665, "rewards/meter/std": 0.4575249254703522, "rewards/total_composite/mean": 0.3376147150993347, "rewards/total_composite/std": 0.22579079866409302, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023648738861084, "sampling/importance_sampling_ratio/min": 0.294707715511322, "sampling/sampling_logp_difference/max": 1.221771240234375, "sampling/sampling_logp_difference/mean": 0.015463379211723804, "step": 510 }, { "clip_ratio/high_max": 0.058790234150364995, "clip_ratio/high_mean": 0.058790234150364995, "clip_ratio/low_mean": 0.03445858974009752, "clip_ratio/low_min": 0.03445858974009752, "clip_ratio/region_mean": 0.09324882389046252, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 63.75, "completions/mean_terminated_length": 63.75, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 1.0178688187152147, "epoch": 0.019732777262897745, "frac_reward_zero_std": 0.0, "grad_norm": 10.033199310302734, "learning_rate": 8.454545454545455e-06, "loss": -0.019, "num_tokens": 1105240.0, "reward": 0.6394338011741638, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.6957893371582031, "reward_meter_std": 0.35945481061935425, "reward_std": 0.35790061950683594, "reward_total_composite_mean": 0.6394338011741638, "reward_total_composite_std": 0.35790061950683594, "reward_total_mean": 0.6394338011741638, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.6957893371582031, "rewards/meter/std": 0.35945481061935425, "rewards/total_composite/mean": 0.6394338011741638, "rewards/total_composite/std": 0.35790061950683594, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0079913139343262, "sampling/importance_sampling_ratio/min": 0.06637737900018692, "sampling/sampling_logp_difference/max": 2.7123990058898926, "sampling/sampling_logp_difference/mean": 0.11444373428821564, "step": 511 }, { "clip_ratio/high_max": 0.014946788433007896, "clip_ratio/high_mean": 0.014946788433007896, "clip_ratio/low_mean": 0.00832630880177021, "clip_ratio/low_min": 0.00832630880177021, "clip_ratio/region_mean": 0.023273097234778106, "completions/clipped_ratio": 0.0, "completions/max_length": 278.0, "completions/max_terminated_length": 278.0, "completions/mean_length": 225.75, "completions/mean_terminated_length": 225.75, "completions/min_length": 186.0, "completions/min_terminated_length": 186.0, "entropy": 0.19403257127851248, "epoch": 0.01977139326536917, "frac_reward_zero_std": 0.0, "grad_norm": 1.5483953952789307, "learning_rate": 8.451515151515151e-06, "loss": -0.0532, "num_tokens": 1108518.0, "reward": 0.8210574388504028, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.8709026575088501, "reward_meter_std": 0.35162675380706787, "reward_std": 0.34322601556777954, "reward_total_composite_mean": 0.8210574388504028, "reward_total_composite_std": 0.34322604537010193, "reward_total_mean": 0.8210574388504028, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.8709026575088501, "rewards/meter/std": 0.35162675380706787, "rewards/total_composite/mean": 0.8210574388504028, "rewards/total_composite/std": 0.34322604537010193, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0026448965072632, "sampling/importance_sampling_ratio/min": 0.3978947103023529, "sampling/sampling_logp_difference/max": 0.9215679168701172, "sampling/sampling_logp_difference/mean": 0.02275705337524414, "step": 512 }, { "clip_ratio/high_max": 0.02126406622119248, "clip_ratio/high_mean": 0.02126406622119248, "clip_ratio/low_mean": 0.009132357081398368, "clip_ratio/low_min": 0.009132357081398368, "clip_ratio/region_mean": 0.030396423302590847, "completions/clipped_ratio": 0.0, "completions/max_length": 175.0, "completions/max_terminated_length": 175.0, "completions/mean_length": 149.25, "completions/mean_terminated_length": 149.25, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.19797090534120798, "epoch": 0.019810009267840593, "frac_reward_zero_std": 0.0, "grad_norm": 2.7554123401641846, "learning_rate": 8.44848484848485e-06, "loss": -0.0103, "num_tokens": 1110976.0, "reward": 0.8653509020805359, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.8965615034103394, "reward_meter_std": 0.14327730238437653, "reward_std": 0.14502419531345367, "reward_total_composite_mean": 0.8653509020805359, "reward_total_composite_std": 0.14502419531345367, "reward_total_mean": 0.8653509020805359, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.8965615034103394, "rewards/meter/std": 0.14327730238437653, "rewards/total_composite/mean": 0.8653509020805359, "rewards/total_composite/std": 0.14502419531345367, "sampling/importance_sampling_ratio/max": 1.710719108581543, "sampling/importance_sampling_ratio/mean": 1.002771258354187, "sampling/importance_sampling_ratio/min": 0.2964406907558441, "sampling/sampling_logp_difference/max": 1.2159080505371094, "sampling/sampling_logp_difference/mean": 0.02919297106564045, "step": 513 }, { "clip_ratio/high_max": 0.051333898678421974, "clip_ratio/high_mean": 0.051333898678421974, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/region_mean": 0.054712277138605714, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 67.75, "completions/mean_terminated_length": 67.75, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.40955257788300514, "epoch": 0.019848625270312018, "frac_reward_zero_std": 0.0, "grad_norm": 8.517362594604492, "learning_rate": 8.445454545454547e-06, "loss": 0.0446, "num_tokens": 1112726.0, "reward": 0.8835674524307251, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8835674524307251, "reward_meter_std": 0.292568564414978, "reward_std": 0.2925685942173004, "reward_total_composite_mean": 0.8835674524307251, "reward_total_composite_std": 0.292568564414978, "reward_total_mean": 0.8835674524307251, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8835674524307251, "rewards/meter/std": 0.292568564414978, "rewards/total_composite/mean": 0.8835674524307251, "rewards/total_composite/std": 0.292568564414978, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.009161114692688, "sampling/importance_sampling_ratio/min": 0.2343650758266449, "sampling/sampling_logp_difference/max": 1.4508752822875977, "sampling/sampling_logp_difference/mean": 0.06897350400686264, "step": 514 }, { "clip_ratio/high_max": 0.043231920688413084, "clip_ratio/high_mean": 0.043231920688413084, "clip_ratio/low_mean": 0.021710526663810015, "clip_ratio/low_min": 0.021710526663810015, "clip_ratio/region_mean": 0.0649424473522231, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 61.625, "completions/mean_terminated_length": 61.625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.4629223048686981, "epoch": 0.01988724127278344, "frac_reward_zero_std": 0.0, "grad_norm": 5.492783069610596, "learning_rate": 8.442424242424243e-06, "loss": -0.0222, "num_tokens": 1114499.0, "reward": 0.7765995264053345, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7765995264053345, "reward_meter_std": 0.4026697874069214, "reward_std": 0.4026697874069214, "reward_total_composite_mean": 0.7765995264053345, "reward_total_composite_std": 0.4026697874069214, "reward_total_mean": 0.7765995264053345, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7765995264053345, "rewards/meter/std": 0.4026697874069214, "rewards/total_composite/mean": 0.7765995264053345, "rewards/total_composite/std": 0.4026697874069214, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0022004842758179, "sampling/importance_sampling_ratio/min": 0.2164708971977234, "sampling/sampling_logp_difference/max": 1.8261196613311768, "sampling/sampling_logp_difference/mean": 0.06085870414972305, "step": 515 }, { "clip_ratio/high_max": 0.07153899129480124, "clip_ratio/high_mean": 0.07153899129480124, "clip_ratio/low_mean": 0.04520996939390898, "clip_ratio/low_min": 0.04520996939390898, "clip_ratio/region_mean": 0.11674896068871021, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 32.625, "completions/mean_terminated_length": 32.625, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 1.2552415579557419, "epoch": 0.019925857275254866, "frac_reward_zero_std": 0.0, "grad_norm": 13.839630126953125, "learning_rate": 8.43939393939394e-06, "loss": -0.0002, "num_tokens": 1116008.0, "reward": 0.7724308371543884, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7724308371543884, "reward_meter_std": 0.31887128949165344, "reward_std": 0.31887128949165344, "reward_total_composite_mean": 0.7724308371543884, "reward_total_composite_std": 0.31887128949165344, "reward_total_mean": 0.7724308371543884, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7724308371543884, "rewards/meter/std": 0.31887128949165344, "rewards/total_composite/mean": 0.7724308371543884, "rewards/total_composite/std": 0.31887128949165344, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0144718885421753, "sampling/importance_sampling_ratio/min": 0.30714622139930725, "sampling/sampling_logp_difference/max": 1.1804313659667969, "sampling/sampling_logp_difference/mean": 0.13659369945526123, "step": 516 }, { "clip_ratio/high_max": 0.0776620376855135, "clip_ratio/high_mean": 0.0776620376855135, "clip_ratio/low_mean": 0.02741228137165308, "clip_ratio/low_min": 0.02741228137165308, "clip_ratio/region_mean": 0.10507431905716658, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 24.875, "completions/mean_terminated_length": 24.875, "completions/min_length": 18.0, "completions/min_terminated_length": 18.0, "entropy": 1.112723309546709, "epoch": 0.01996447327772629, "frac_reward_zero_std": 0.0, "grad_norm": 17.498676300048828, "learning_rate": 8.436363636363637e-06, "loss": -0.0974, "num_tokens": 1117535.0, "reward": 0.9236002564430237, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9236002564430237, "reward_meter_std": 0.16382306814193726, "reward_std": 0.16382306814193726, "reward_total_composite_mean": 0.9236002564430237, "reward_total_composite_std": 0.16382306814193726, "reward_total_mean": 0.9236002564430237, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9236002564430237, "rewards/meter/std": 0.16382306814193726, "rewards/total_composite/mean": 0.9236002564430237, "rewards/total_composite/std": 0.16382306814193726, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0398297309875488, "sampling/importance_sampling_ratio/min": 0.27661052346229553, "sampling/sampling_logp_difference/max": 1.2851448059082031, "sampling/sampling_logp_difference/mean": 0.13017626106739044, "step": 517 }, { "clip_ratio/high_max": 0.049344516824930906, "clip_ratio/high_mean": 0.049344516824930906, "clip_ratio/low_mean": 0.007462686393409967, "clip_ratio/low_min": 0.007462686393409967, "clip_ratio/region_mean": 0.056807203218340874, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 61.75, "completions/mean_terminated_length": 61.75, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.3643992841243744, "epoch": 0.020003089280197714, "frac_reward_zero_std": 0.0, "grad_norm": 15.10335922241211, "learning_rate": 8.433333333333334e-06, "loss": 0.047, "num_tokens": 1119517.0, "reward": 0.9054743051528931, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9054743051528931, "reward_meter_std": 0.21583297848701477, "reward_std": 0.21583294868469238, "reward_total_composite_mean": 0.9054743051528931, "reward_total_composite_std": 0.21583297848701477, "reward_total_mean": 0.9054743051528931, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9054743051528931, "rewards/meter/std": 0.21583297848701477, "rewards/total_composite/mean": 0.9054743051528931, "rewards/total_composite/std": 0.21583297848701477, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0084285736083984, "sampling/importance_sampling_ratio/min": 0.18905863165855408, "sampling/sampling_logp_difference/max": 1.6656980514526367, "sampling/sampling_logp_difference/mean": 0.07659897208213806, "step": 518 }, { "clip_ratio/high_max": 0.04268710082396865, "clip_ratio/high_mean": 0.04268710082396865, "clip_ratio/low_mean": 0.007263681618496776, "clip_ratio/low_min": 0.007263681618496776, "clip_ratio/region_mean": 0.049950782442465425, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 70.0, "completions/mean_terminated_length": 70.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.3383452221751213, "epoch": 0.020041705282669138, "frac_reward_zero_std": 0.0, "grad_norm": 8.149646759033203, "learning_rate": 8.43030303030303e-06, "loss": 0.0138, "num_tokens": 1121333.0, "reward": 0.9806559085845947, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9806559085845947, "reward_meter_std": 0.02947128564119339, "reward_std": 0.029471300542354584, "reward_total_composite_mean": 0.9806559085845947, "reward_total_composite_std": 0.02947128564119339, "reward_total_mean": 0.9806559085845947, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9806559085845947, "rewards/meter/std": 0.02947128564119339, "rewards/total_composite/mean": 0.9806559085845947, "rewards/total_composite/std": 0.02947128564119339, "sampling/importance_sampling_ratio/max": 1.8948339223861694, "sampling/importance_sampling_ratio/mean": 1.0027549266815186, "sampling/importance_sampling_ratio/min": 0.009218051098287106, "sampling/sampling_logp_difference/max": 4.686591625213623, "sampling/sampling_logp_difference/mean": 0.07023530453443527, "step": 519 }, { "clip_ratio/high_max": 0.07098443387076259, "clip_ratio/high_mean": 0.07098443387076259, "clip_ratio/low_mean": 0.007692307699471712, "clip_ratio/low_min": 0.007692307699471712, "clip_ratio/region_mean": 0.0786767415702343, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 55.75, "completions/mean_terminated_length": 55.75, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.4810672476887703, "epoch": 0.020080321285140562, "frac_reward_zero_std": 0.0, "grad_norm": 8.837297439575195, "learning_rate": 8.427272727272729e-06, "loss": 0.0682, "num_tokens": 1123139.0, "reward": 0.8729170560836792, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8729170560836792, "reward_meter_std": 0.29740720987319946, "reward_std": 0.2974071800708771, "reward_total_composite_mean": 0.8729170560836792, "reward_total_composite_std": 0.29740720987319946, "reward_total_mean": 0.8729170560836792, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8729170560836792, "rewards/meter/std": 0.29740720987319946, "rewards/total_composite/mean": 0.8729170560836792, "rewards/total_composite/std": 0.29740720987319946, "sampling/importance_sampling_ratio/max": 1.8521945476531982, "sampling/importance_sampling_ratio/mean": 0.9940183162689209, "sampling/importance_sampling_ratio/min": 0.2821709215641022, "sampling/sampling_logp_difference/max": 1.265242338180542, "sampling/sampling_logp_difference/mean": 0.07541073858737946, "step": 520 }, { "clip_ratio/high_max": 0.022381525253877044, "clip_ratio/high_mean": 0.022381525253877044, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/region_mean": 0.03071485902182758, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 59.75, "completions/mean_terminated_length": 59.75, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.1859533153474331, "epoch": 0.020118937287611986, "frac_reward_zero_std": 0.0, "grad_norm": 16.822296142578125, "learning_rate": 8.424242424242425e-06, "loss": 0.0135, "num_tokens": 1124881.0, "reward": 0.9238840341567993, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9238840341567993, "reward_meter_std": 0.19443634152412415, "reward_std": 0.19443634152412415, "reward_total_composite_mean": 0.9238840341567993, "reward_total_composite_std": 0.19443634152412415, "reward_total_mean": 0.9238840341567993, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9238840341567993, "rewards/meter/std": 0.19443634152412415, "rewards/total_composite/mean": 0.9238840341567993, "rewards/total_composite/std": 0.19443634152412415, "sampling/importance_sampling_ratio/max": 1.7410789728164673, "sampling/importance_sampling_ratio/mean": 0.9970530867576599, "sampling/importance_sampling_ratio/min": 0.04523586481809616, "sampling/sampling_logp_difference/max": 3.09586501121521, "sampling/sampling_logp_difference/mean": 0.039955347776412964, "step": 521 }, { "clip_ratio/high_max": 0.028243856504559517, "clip_ratio/high_mean": 0.028243856504559517, "clip_ratio/low_mean": 0.01347197126597166, "clip_ratio/low_min": 0.01347197126597166, "clip_ratio/region_mean": 0.04171582777053118, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 75.125, "completions/mean_terminated_length": 75.125, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.3937871027737856, "epoch": 0.02015755329008341, "frac_reward_zero_std": 0.0, "grad_norm": 5.966022491455078, "learning_rate": 8.421212121212122e-06, "loss": 0.0201, "num_tokens": 1126658.0, "reward": 0.9940762519836426, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9940762519836426, "reward_meter_std": 0.0017088382737711072, "reward_std": 0.0017088415334001184, "reward_total_composite_mean": 0.9940762519836426, "reward_total_composite_std": 0.0017088382737711072, "reward_total_mean": 0.9940762519836426, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9940762519836426, "rewards/meter/std": 0.0017088382737711072, "rewards/total_composite/mean": 0.9940762519836426, "rewards/total_composite/std": 0.0017088382737711072, "sampling/importance_sampling_ratio/max": 1.9666850566864014, "sampling/importance_sampling_ratio/mean": 1.0083460807800293, "sampling/importance_sampling_ratio/min": 0.3674972355365753, "sampling/sampling_logp_difference/max": 1.0010395050048828, "sampling/sampling_logp_difference/mean": 0.05746918171644211, "step": 522 }, { "clip_ratio/high_max": 0.04396977473516017, "clip_ratio/high_mean": 0.04396977473516017, "clip_ratio/low_mean": 0.004464285913854837, "clip_ratio/low_min": 0.004464285913854837, "clip_ratio/region_mean": 0.04843406064901501, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 62.75, "completions/mean_terminated_length": 62.75, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.23601900041103363, "epoch": 0.020196169292554834, "frac_reward_zero_std": 0.0, "grad_norm": 4.775942802429199, "learning_rate": 8.418181818181819e-06, "loss": -0.0272, "num_tokens": 1128392.0, "reward": 0.8802205324172974, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8802205324172974, "reward_meter_std": 0.31313174962997437, "reward_std": 0.31313174962997437, "reward_total_composite_mean": 0.8802205324172974, "reward_total_composite_std": 0.31313174962997437, "reward_total_mean": 0.8802205324172974, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8802205324172974, "rewards/meter/std": 0.31313174962997437, "rewards/total_composite/mean": 0.8802205324172974, "rewards/total_composite/std": 0.31313174962997437, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9997758269309998, "sampling/importance_sampling_ratio/min": 0.22748437523841858, "sampling/sampling_logp_difference/max": 1.4806737899780273, "sampling/sampling_logp_difference/mean": 0.040101900696754456, "step": 523 }, { "clip_ratio/high_max": 0.0902106543071568, "clip_ratio/high_mean": 0.0902106543071568, "clip_ratio/low_mean": 0.0262486576102674, "clip_ratio/low_min": 0.0262486576102674, "clip_ratio/region_mean": 0.1164593119174242, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 62.125, "completions/mean_terminated_length": 62.125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 1.2009354755282402, "epoch": 0.02023478529502626, "frac_reward_zero_std": 0.0, "grad_norm": 8.702469825744629, "learning_rate": 8.415151515151516e-06, "loss": -0.0291, "num_tokens": 1130217.0, "reward": 0.8824061155319214, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8824061155319214, "reward_meter_std": 0.2138787806034088, "reward_std": 0.2138787806034088, "reward_total_composite_mean": 0.8824061155319214, "reward_total_composite_std": 0.2138787806034088, "reward_total_mean": 0.8824061155319214, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8824061155319214, "rewards/meter/std": 0.2138787806034088, "rewards/total_composite/mean": 0.8824061155319214, "rewards/total_composite/std": 0.2138787806034088, "sampling/importance_sampling_ratio/max": 1.9018948078155518, "sampling/importance_sampling_ratio/mean": 1.0184638500213623, "sampling/importance_sampling_ratio/min": 0.21237356960773468, "sampling/sampling_logp_difference/max": 1.5494084358215332, "sampling/sampling_logp_difference/mean": 0.10073237121105194, "step": 524 }, { "clip_ratio/high_max": 0.027191198198124766, "clip_ratio/high_mean": 0.027191198198124766, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/region_mean": 0.031037352047860622, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 59.0, "completions/mean_terminated_length": 59.0, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.3998655714094639, "epoch": 0.020273401297497683, "frac_reward_zero_std": 0.0, "grad_norm": 6.02722692489624, "learning_rate": 8.412121212121212e-06, "loss": 0.0389, "num_tokens": 1131849.0, "reward": 0.9700571298599243, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9700571298599243, "reward_meter_std": 0.05985637754201889, "reward_std": 0.059856388717889786, "reward_total_composite_mean": 0.9700571298599243, "reward_total_composite_std": 0.05985637754201889, "reward_total_mean": 0.9700571298599243, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9700571298599243, "rewards/meter/std": 0.05985637754201889, "rewards/total_composite/mean": 0.9700571298599243, "rewards/total_composite/std": 0.05985637754201889, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01723313331604, "sampling/importance_sampling_ratio/min": 0.3532440662384033, "sampling/sampling_logp_difference/max": 1.0405960083007812, "sampling/sampling_logp_difference/mean": 0.05891498550772667, "step": 525 }, { "clip_ratio/high_max": 0.029155581374652684, "clip_ratio/high_mean": 0.029155581374652684, "clip_ratio/low_mean": 0.021122918464243412, "clip_ratio/low_min": 0.021122918464243412, "clip_ratio/region_mean": 0.050278499838896096, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 70.5, "completions/mean_terminated_length": 70.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.4445030000060797, "epoch": 0.020312017299969107, "frac_reward_zero_std": 0.0, "grad_norm": 5.217512607574463, "learning_rate": 8.40909090909091e-06, "loss": 0.0231, "num_tokens": 1133805.0, "reward": 0.9931546449661255, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9931546449661255, "reward_meter_std": 0.004212968982756138, "reward_std": 0.00421295827254653, "reward_total_composite_mean": 0.9931546449661255, "reward_total_composite_std": 0.004212968982756138, "reward_total_mean": 0.9931546449661255, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9931546449661255, "rewards/meter/std": 0.004212968982756138, "rewards/total_composite/mean": 0.9931546449661255, "rewards/total_composite/std": 0.004212968982756138, "sampling/importance_sampling_ratio/max": 1.9715811014175415, "sampling/importance_sampling_ratio/mean": 1.0042046308517456, "sampling/importance_sampling_ratio/min": 0.214838445186615, "sampling/sampling_logp_difference/max": 1.5378689765930176, "sampling/sampling_logp_difference/mean": 0.0634373277425766, "step": 526 }, { "clip_ratio/high_max": 0.016355944564566016, "clip_ratio/high_mean": 0.016355944564566016, "clip_ratio/low_mean": 0.01564345066435635, "clip_ratio/low_min": 0.01564345066435635, "clip_ratio/region_mean": 0.03199939522892237, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 75.375, "completions/mean_terminated_length": 75.375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.3602394051849842, "epoch": 0.02035063330244053, "frac_reward_zero_std": 0.0, "grad_norm": 2.7269952297210693, "learning_rate": 8.406060606060606e-06, "loss": 0.006, "num_tokens": 1135576.0, "reward": 0.9965004920959473, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9965004920959473, "reward_meter_std": 0.0009321427205577493, "reward_std": 0.000932130089495331, "reward_total_composite_mean": 0.9965004920959473, "reward_total_composite_std": 0.0009321427205577493, "reward_total_mean": 0.9965004920959473, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9965004920959473, "rewards/meter/std": 0.0009321427205577493, "rewards/total_composite/mean": 0.9965004920959473, "rewards/total_composite/std": 0.0009321427205577493, "sampling/importance_sampling_ratio/max": 1.7970446348190308, "sampling/importance_sampling_ratio/mean": 1.0171750783920288, "sampling/importance_sampling_ratio/min": 0.4395350217819214, "sampling/sampling_logp_difference/max": 0.822037935256958, "sampling/sampling_logp_difference/mean": 0.04060669243335724, "step": 527 }, { "clip_ratio/high_max": 0.07623076229356229, "clip_ratio/high_mean": 0.07623076229356229, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.08001864119432867, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.784766998142004, "epoch": 0.020389249304911955, "frac_reward_zero_std": 0.0, "grad_norm": 9.207352638244629, "learning_rate": 8.403030303030304e-06, "loss": 0.0118, "num_tokens": 1137326.0, "reward": 0.9670517444610596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9670517444610596, "reward_meter_std": 0.06938152760267258, "reward_std": 0.06938153505325317, "reward_total_composite_mean": 0.9670517444610596, "reward_total_composite_std": 0.06938152760267258, "reward_total_mean": 0.9670517444610596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9670517444610596, "rewards/meter/std": 0.06938152760267258, "rewards/total_composite/mean": 0.9670517444610596, "rewards/total_composite/std": 0.06938152760267258, "sampling/importance_sampling_ratio/max": 1.738789677619934, "sampling/importance_sampling_ratio/mean": 0.9962921738624573, "sampling/importance_sampling_ratio/min": 0.26605382561683655, "sampling/sampling_logp_difference/max": 1.324056625366211, "sampling/sampling_logp_difference/mean": 0.08710537105798721, "step": 528 }, { "clip_ratio/high_max": 0.015567422262392938, "clip_ratio/high_mean": 0.015567422262392938, "clip_ratio/low_mean": 0.029255319386720657, "clip_ratio/low_min": 0.029255319386720657, "clip_ratio/region_mean": 0.044822741649113595, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.30135350674390793, "epoch": 0.02042786530738338, "frac_reward_zero_std": 0.0, "grad_norm": 7.656765460968018, "learning_rate": 8.400000000000001e-06, "loss": -0.0752, "num_tokens": 1139126.0, "reward": 0.9007197618484497, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9007197618484497, "reward_meter_std": 0.2582366466522217, "reward_std": 0.2582366466522217, "reward_total_composite_mean": 0.9007197618484497, "reward_total_composite_std": 0.2582366466522217, "reward_total_mean": 0.9007197618484497, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9007197618484497, "rewards/meter/std": 0.2582366466522217, "rewards/total_composite/mean": 0.9007197618484497, "rewards/total_composite/std": 0.2582366466522217, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0085376501083374, "sampling/importance_sampling_ratio/min": 0.2871541976928711, "sampling/sampling_logp_difference/max": 1.2477359771728516, "sampling/sampling_logp_difference/mean": 0.03346162289381027, "step": 529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.020466481309854803, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 8.396969696969698e-06, "loss": 0.0, "num_tokens": 1140846.0, "reward": 0.7890878915786743, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7941176891326904, "reward_count_adherence_std": 0.03144249692559242, "reward_meter_mean": 0.9937265515327454, "reward_meter_std": 0.002974079456180334, "reward_std": 0.029924683272838593, "reward_total_composite_mean": 0.7890878915786743, "reward_total_composite_std": 0.029924675822257996, "reward_total_mean": 0.7890878915786743, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7941176891326904, "rewards/count_adherence/std": 0.03144249692559242, "rewards/meter/mean": 0.9937265515327454, "rewards/meter/std": 0.002974079456180334, "rewards/total_composite/mean": 0.7890878915786743, "rewards/total_composite/std": 0.029924675822257996, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 530 }, { "clip_ratio/high_max": 0.0648979377001524, "clip_ratio/high_mean": 0.0648979377001524, "clip_ratio/low_mean": 0.08407607581466436, "clip_ratio/low_min": 0.08407607581466436, "clip_ratio/region_mean": 0.14897401351481676, "completions/clipped_ratio": 0.0, "completions/max_length": 53.0, "completions/max_terminated_length": 53.0, "completions/mean_length": 44.375, "completions/mean_terminated_length": 44.375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 1.1344088204205036, "epoch": 0.020505097312326227, "frac_reward_zero_std": 0.0, "grad_norm": 15.459403038024902, "learning_rate": 8.393939393939394e-06, "loss": 0.0262, "num_tokens": 1142401.0, "reward": 0.564997673034668, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.564997673034668, "reward_meter_std": 0.3845665752887726, "reward_std": 0.3845665454864502, "reward_total_composite_mean": 0.564997673034668, "reward_total_composite_std": 0.3845665752887726, "reward_total_mean": 0.564997673034668, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.564997673034668, "rewards/meter/std": 0.3845665752887726, "rewards/total_composite/mean": 0.564997673034668, "rewards/total_composite/std": 0.3845665752887726, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0244461297988892, "sampling/importance_sampling_ratio/min": 0.09556441009044647, "sampling/sampling_logp_difference/max": 2.347954750061035, "sampling/sampling_logp_difference/mean": 0.17102597653865814, "step": 531 }, { "clip_ratio/high_max": 0.0692659798078239, "clip_ratio/high_mean": 0.0692659798078239, "clip_ratio/low_mean": 0.01616688398644328, "clip_ratio/low_min": 0.01616688398644328, "clip_ratio/region_mean": 0.08543286379426718, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 60.375, "completions/mean_terminated_length": 60.375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.5591697357594967, "epoch": 0.02054371331479765, "frac_reward_zero_std": 0.0, "grad_norm": 11.56308364868164, "learning_rate": 8.390909090909091e-06, "loss": 0.0452, "num_tokens": 1144260.0, "reward": 0.772533655166626, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.772533655166626, "reward_meter_std": 0.38782942295074463, "reward_std": 0.38782939314842224, "reward_total_composite_mean": 0.772533655166626, "reward_total_composite_std": 0.38782942295074463, "reward_total_mean": 0.772533655166626, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.772533655166626, "rewards/meter/std": 0.38782942295074463, "rewards/total_composite/mean": 0.772533655166626, "rewards/total_composite/std": 0.38782942295074463, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002143383026123, "sampling/importance_sampling_ratio/min": 0.16826894879341125, "sampling/sampling_logp_difference/max": 1.7821917533874512, "sampling/sampling_logp_difference/mean": 0.08298543095588684, "step": 532 }, { "clip_ratio/high_max": 0.059228118509054184, "clip_ratio/high_mean": 0.059228118509054184, "clip_ratio/low_mean": 0.02431722730398178, "clip_ratio/low_min": 0.02431722730398178, "clip_ratio/region_mean": 0.08354534581303596, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 31.75, "completions/mean_terminated_length": 31.75, "completions/min_length": 22.0, "completions/min_terminated_length": 22.0, "entropy": 0.6379514243453741, "epoch": 0.020582329317269075, "frac_reward_zero_std": 0.0, "grad_norm": 7.510979652404785, "learning_rate": 8.387878787878788e-06, "loss": -0.0447, "num_tokens": 1145610.0, "reward": 0.994328498840332, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994328498840332, "reward_meter_std": 0.00353585509583354, "reward_std": 0.00353585509583354, "reward_total_composite_mean": 0.994328498840332, "reward_total_composite_std": 0.00353585509583354, "reward_total_mean": 0.994328498840332, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994328498840332, "rewards/meter/std": 0.00353585509583354, "rewards/total_composite/mean": 0.994328498840332, "rewards/total_composite/std": 0.00353585509583354, "sampling/importance_sampling_ratio/max": 1.9707249402999878, "sampling/importance_sampling_ratio/mean": 0.9933980703353882, "sampling/importance_sampling_ratio/min": 0.33012497425079346, "sampling/sampling_logp_difference/max": 1.1082839965820312, "sampling/sampling_logp_difference/mean": 0.07902076840400696, "step": 533 }, { "clip_ratio/high_max": 0.00898129097186029, "clip_ratio/high_mean": 0.00898129097186029, "clip_ratio/low_mean": 0.008662280859425664, "clip_ratio/low_min": 0.008662280859425664, "clip_ratio/region_mean": 0.017643571831285954, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 56.25, "completions/mean_terminated_length": 56.25, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.17711176443845034, "epoch": 0.0206209453197405, "frac_reward_zero_std": 0.0, "grad_norm": 4.608026504516602, "learning_rate": 8.384848484848485e-06, "loss": 0.0332, "num_tokens": 1147340.0, "reward": 0.9940440058708191, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9940440058708191, "reward_meter_std": 0.0018207214307039976, "reward_std": 0.0018207177054136992, "reward_total_composite_mean": 0.9940440058708191, "reward_total_composite_std": 0.0018207214307039976, "reward_total_mean": 0.9940440058708191, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9940440058708191, "rewards/meter/std": 0.0018207214307039976, "rewards/total_composite/mean": 0.9940440058708191, "rewards/total_composite/std": 0.0018207214307039976, "sampling/importance_sampling_ratio/max": 1.6200429201126099, "sampling/importance_sampling_ratio/mean": 1.0104416608810425, "sampling/importance_sampling_ratio/min": 0.3822028934955597, "sampling/sampling_logp_difference/max": 0.9618037343025208, "sampling/sampling_logp_difference/mean": 0.02367156185209751, "step": 534 }, { "clip_ratio/high_max": 0.02654118835926056, "clip_ratio/high_mean": 0.02654118835926056, "clip_ratio/low_mean": 0.008223684038966894, "clip_ratio/low_min": 0.008223684038966894, "clip_ratio/region_mean": 0.03476487239822745, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 66.25, "completions/mean_terminated_length": 66.25, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.3021476771682501, "epoch": 0.020659561322211924, "frac_reward_zero_std": 0.0, "grad_norm": 5.5635881423950195, "learning_rate": 8.381818181818183e-06, "loss": 0.0063, "num_tokens": 1149134.0, "reward": 0.7755165100097656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7755165100097656, "reward_meter_std": 0.3405035734176636, "reward_std": 0.3405035436153412, "reward_total_composite_mean": 0.7755165100097656, "reward_total_composite_std": 0.3405035734176636, "reward_total_mean": 0.7755165100097656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7755165100097656, "rewards/meter/std": 0.3405035734176636, "rewards/total_composite/mean": 0.7755165100097656, "rewards/total_composite/std": 0.3405035734176636, "sampling/importance_sampling_ratio/max": 1.6945619583129883, "sampling/importance_sampling_ratio/mean": 1.0043402910232544, "sampling/importance_sampling_ratio/min": 0.16238459944725037, "sampling/sampling_logp_difference/max": 1.8177876472473145, "sampling/sampling_logp_difference/mean": 0.044020526111125946, "step": 535 }, { "clip_ratio/high_max": 0.053157048765569925, "clip_ratio/high_mean": 0.053157048765569925, "clip_ratio/low_mean": 0.043238147161901, "clip_ratio/low_min": 0.043238147161901, "clip_ratio/region_mean": 0.09639519592747092, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 53.75, "completions/mean_terminated_length": 53.75, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 1.1170207932591438, "epoch": 0.020698177324683348, "frac_reward_zero_std": 0.0, "grad_norm": 8.588635444641113, "learning_rate": 8.37878787878788e-06, "loss": -0.0758, "num_tokens": 1150740.0, "reward": 0.7046811580657959, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7046811580657959, "reward_meter_std": 0.4017917513847351, "reward_std": 0.4017917513847351, "reward_total_composite_mean": 0.7046811580657959, "reward_total_composite_std": 0.4017917513847351, "reward_total_mean": 0.7046811580657959, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7046811580657959, "rewards/meter/std": 0.4017917513847351, "rewards/total_composite/mean": 0.7046811580657959, "rewards/total_composite/std": 0.4017917513847351, "sampling/importance_sampling_ratio/max": 1.8760080337524414, "sampling/importance_sampling_ratio/mean": 1.0192784070968628, "sampling/importance_sampling_ratio/min": 0.31289422512054443, "sampling/sampling_logp_difference/max": 1.1618900299072266, "sampling/sampling_logp_difference/mean": 0.10004501044750214, "step": 536 }, { "clip_ratio/high_max": 0.019980506971478462, "clip_ratio/high_mean": 0.019980506971478462, "clip_ratio/low_mean": 0.03469064529053867, "clip_ratio/low_min": 0.03469064529053867, "clip_ratio/region_mean": 0.05467115226201713, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 49.75, "completions/mean_terminated_length": 49.75, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.6782696768641472, "epoch": 0.020736793327154772, "frac_reward_zero_std": 0.0, "grad_norm": 8.848946571350098, "learning_rate": 8.375757575757576e-06, "loss": -0.0426, "num_tokens": 1152266.0, "reward": 0.9934061765670776, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9934061765670776, "reward_meter_std": 0.0037812571972608566, "reward_std": 0.003781270468607545, "reward_total_composite_mean": 0.9934061765670776, "reward_total_composite_std": 0.0037812571972608566, "reward_total_mean": 0.9934061765670776, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9934061765670776, "rewards/meter/std": 0.0037812571972608566, "rewards/total_composite/mean": 0.9934061765670776, "rewards/total_composite/std": 0.0037812571972608566, "sampling/importance_sampling_ratio/max": 1.9671330451965332, "sampling/importance_sampling_ratio/mean": 1.0079725980758667, "sampling/importance_sampling_ratio/min": 0.20439153909683228, "sampling/sampling_logp_difference/max": 1.5877177715301514, "sampling/sampling_logp_difference/mean": 0.06811977922916412, "step": 537 }, { "clip_ratio/high_max": 0.026057497365400195, "clip_ratio/high_mean": 0.026057497365400195, "clip_ratio/low_mean": 0.016544118523597717, "clip_ratio/low_min": 0.016544118523597717, "clip_ratio/region_mean": 0.04260161588899791, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 60.625, "completions/mean_terminated_length": 60.625, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.5225202403962612, "epoch": 0.020775409329626196, "frac_reward_zero_std": 0.0, "grad_norm": 11.105710983276367, "learning_rate": 8.372727272727273e-06, "loss": 0.0562, "num_tokens": 1153991.0, "reward": 0.9464359283447266, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9464359283447266, "reward_meter_std": 0.10982144623994827, "reward_std": 0.10982143878936768, "reward_total_composite_mean": 0.9464359283447266, "reward_total_composite_std": 0.10982144623994827, "reward_total_mean": 0.9464359283447266, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9464359283447266, "rewards/meter/std": 0.10982144623994827, "rewards/total_composite/mean": 0.9464359283447266, "rewards/total_composite/std": 0.10982144623994827, "sampling/importance_sampling_ratio/max": 1.6170642375946045, "sampling/importance_sampling_ratio/mean": 1.0101873874664307, "sampling/importance_sampling_ratio/min": 2.127368787796513e-07, "sampling/sampling_logp_difference/max": 15.36320972442627, "sampling/sampling_logp_difference/mean": 0.08471371978521347, "step": 538 }, { "clip_ratio/high_max": 0.05073056067340076, "clip_ratio/high_mean": 0.05073056067340076, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/region_mean": 0.05641237902455032, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 63.875, "completions/mean_terminated_length": 63.875, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.5723963566124439, "epoch": 0.02081402533209762, "frac_reward_zero_std": 0.0, "grad_norm": 3.6939918994903564, "learning_rate": 8.36969696969697e-06, "loss": 0.0146, "num_tokens": 1155878.0, "reward": 0.9885964393615723, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9885964393615723, "reward_meter_std": 0.016855819150805473, "reward_std": 0.01685582473874092, "reward_total_composite_mean": 0.9885964393615723, "reward_total_composite_std": 0.016855819150805473, "reward_total_mean": 0.9885964393615723, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9885964393615723, "rewards/meter/std": 0.016855819150805473, "rewards/total_composite/mean": 0.9885964393615723, "rewards/total_composite/std": 0.016855819150805473, "sampling/importance_sampling_ratio/max": 1.6128665208816528, "sampling/importance_sampling_ratio/mean": 1.0144485235214233, "sampling/importance_sampling_ratio/min": 0.3828132748603821, "sampling/sampling_logp_difference/max": 0.9602079391479492, "sampling/sampling_logp_difference/mean": 0.04388084262609482, "step": 539 }, { "clip_ratio/high_max": 0.07365514896810055, "clip_ratio/high_mean": 0.07365514896810055, "clip_ratio/low_mean": 0.008426966145634651, "clip_ratio/low_min": 0.008426966145634651, "clip_ratio/region_mean": 0.0820821151137352, "completions/clipped_ratio": 0.0, "completions/max_length": 89.0, "completions/max_terminated_length": 89.0, "completions/mean_length": 63.625, "completions/mean_terminated_length": 63.625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.48294258303940296, "epoch": 0.020852641334569044, "frac_reward_zero_std": 0.0, "grad_norm": 12.970458030700684, "learning_rate": 8.366666666666667e-06, "loss": 0.1671, "num_tokens": 1157571.0, "reward": 0.9218109250068665, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9840399622917175, "reward_meter_std": 0.00786462053656578, "reward_std": 0.1714293360710144, "reward_total_composite_mean": 0.9218109250068665, "reward_total_composite_std": 0.1714293360710144, "reward_total_mean": 0.9218109250068665, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9840399622917175, "rewards/meter/std": 0.00786462053656578, "rewards/total_composite/mean": 0.9218109250068665, "rewards/total_composite/std": 0.1714293360710144, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0064505338668823, "sampling/importance_sampling_ratio/min": 0.3767927885055542, "sampling/sampling_logp_difference/max": 0.9760599136352539, "sampling/sampling_logp_difference/mean": 0.07498990744352341, "step": 540 }, { "clip_ratio/high_max": 0.020859031821601093, "clip_ratio/high_mean": 0.020859031821601093, "clip_ratio/low_mean": 0.02023959648795426, "clip_ratio/low_min": 0.02023959648795426, "clip_ratio/region_mean": 0.04109862830955535, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 62.375, "completions/mean_terminated_length": 62.375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.2864169627428055, "epoch": 0.02089125733704047, "frac_reward_zero_std": 0.0, "grad_norm": 5.643881797790527, "learning_rate": 8.363636363636365e-06, "loss": -0.0019, "num_tokens": 1159414.0, "reward": 0.9919471740722656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9919471740722656, "reward_meter_std": 0.005765520967543125, "reward_std": 0.005765520967543125, "reward_total_composite_mean": 0.9919471740722656, "reward_total_composite_std": 0.005765520967543125, "reward_total_mean": 0.9919471740722656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9919471740722656, "rewards/meter/std": 0.005765520967543125, "rewards/total_composite/mean": 0.9919471740722656, "rewards/total_composite/std": 0.005765520967543125, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0133535861968994, "sampling/importance_sampling_ratio/min": 0.2700084447860718, "sampling/sampling_logp_difference/max": 1.3093020915985107, "sampling/sampling_logp_difference/mean": 0.046641286462545395, "step": 541 }, { "clip_ratio/high_max": 0.01289307966362685, "clip_ratio/high_mean": 0.01289307966362685, "clip_ratio/low_mean": 0.004344935878179967, "clip_ratio/low_min": 0.004344935878179967, "clip_ratio/region_mean": 0.017238015541806817, "completions/clipped_ratio": 0.0, "completions/max_length": 177.0, "completions/max_terminated_length": 177.0, "completions/mean_length": 153.25, "completions/mean_terminated_length": 153.25, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.2024863502010703, "epoch": 0.020929873339511892, "frac_reward_zero_std": 0.0, "grad_norm": 2.8404381275177, "learning_rate": 8.360606060606062e-06, "loss": 0.0263, "num_tokens": 1162016.0, "reward": 0.9956228137016296, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956228137016296, "reward_meter_std": 0.0026213540695607662, "reward_std": 0.002621352905407548, "reward_total_composite_mean": 0.9956228137016296, "reward_total_composite_std": 0.0026213540695607662, "reward_total_mean": 0.9956228137016296, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956228137016296, "rewards/meter/std": 0.0026213540695607662, "rewards/total_composite/mean": 0.9956228137016296, "rewards/total_composite/std": 0.0026213540695607662, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0009527206420898, "sampling/importance_sampling_ratio/min": 0.28226134181022644, "sampling/sampling_logp_difference/max": 1.2649219036102295, "sampling/sampling_logp_difference/mean": 0.026027433574199677, "step": 542 }, { "clip_ratio/high_max": 0.008265212236437947, "clip_ratio/high_mean": 0.008265212236437947, "clip_ratio/low_mean": 0.0014523655408993363, "clip_ratio/low_min": 0.0014523655408993363, "clip_ratio/region_mean": 0.009717577777337283, "completions/clipped_ratio": 0.0, "completions/max_length": 463.0, "completions/max_terminated_length": 463.0, "completions/mean_length": 390.875, "completions/mean_terminated_length": 390.875, "completions/min_length": 335.0, "completions/min_terminated_length": 335.0, "entropy": 0.0655752515885979, "epoch": 0.020968489341983317, "frac_reward_zero_std": 0.0, "grad_norm": 3.049520254135132, "learning_rate": 8.357575757575759e-06, "loss": 0.0722, "num_tokens": 1166687.0, "reward": 0.3688773512840271, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.453125, "reward_count_adherence_std": 0.16280877590179443, "reward_meter_mean": 0.867215096950531, "reward_meter_std": 0.3118921220302582, "reward_std": 0.16893039643764496, "reward_total_composite_mean": 0.3688773512840271, "reward_total_composite_std": 0.16893041133880615, "reward_total_mean": 0.3688773512840271, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.453125, "rewards/count_adherence/std": 0.16280877590179443, "rewards/meter/mean": 0.867215096950531, "rewards/meter/std": 0.3118921220302582, "rewards/total_composite/mean": 0.3688773512840271, "rewards/total_composite/std": 0.16893041133880615, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0000622272491455, "sampling/importance_sampling_ratio/min": 0.30091166496276855, "sampling/sampling_logp_difference/max": 1.2009385824203491, "sampling/sampling_logp_difference/mean": 0.011823274195194244, "step": 543 }, { "clip_ratio/high_max": 0.019809147692285478, "clip_ratio/high_mean": 0.019809147692285478, "clip_ratio/low_mean": 0.035438207909464836, "clip_ratio/low_min": 0.035438207909464836, "clip_ratio/region_mean": 0.055247355601750314, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 95.25, "completions/mean_terminated_length": 95.25, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.2810734435915947, "epoch": 0.02100710534445474, "frac_reward_zero_std": 0.0, "grad_norm": 5.046764850616455, "learning_rate": 8.354545454545455e-06, "loss": -0.0199, "num_tokens": 1168769.0, "reward": 0.9941377639770508, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9941377639770508, "reward_meter_std": 0.004864970687776804, "reward_std": 0.00486496277153492, "reward_total_composite_mean": 0.9941377639770508, "reward_total_composite_std": 0.004864970687776804, "reward_total_mean": 0.9941377639770508, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9941377639770508, "rewards/meter/std": 0.004864970687776804, "rewards/total_composite/mean": 0.9941377639770508, "rewards/total_composite/std": 0.004864970687776804, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0006630420684814, "sampling/importance_sampling_ratio/min": 0.10804898291826248, "sampling/sampling_logp_difference/max": 2.225170612335205, "sampling/sampling_logp_difference/mean": 0.05243554711341858, "step": 544 }, { "clip_ratio/high_max": 0.018268492887727916, "clip_ratio/high_mean": 0.018268492887727916, "clip_ratio/low_mean": 0.004854368977248669, "clip_ratio/low_min": 0.004854368977248669, "clip_ratio/region_mean": 0.023122861864976585, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 98.75, "completions/mean_terminated_length": 98.75, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.23292037099599838, "epoch": 0.021045721346926165, "frac_reward_zero_std": 0.0, "grad_norm": 5.207396030426025, "learning_rate": 8.351515151515152e-06, "loss": 0.0556, "num_tokens": 1170871.0, "reward": 0.9043634533882141, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9043634533882141, "reward_meter_std": 0.18687696754932404, "reward_std": 0.18687698245048523, "reward_total_composite_mean": 0.9043634533882141, "reward_total_composite_std": 0.18687696754932404, "reward_total_mean": 0.9043634533882141, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9043634533882141, "rewards/meter/std": 0.18687696754932404, "rewards/total_composite/mean": 0.9043634533882141, "rewards/total_composite/std": 0.18687696754932404, "sampling/importance_sampling_ratio/max": 1.8148711919784546, "sampling/importance_sampling_ratio/mean": 1.0059583187103271, "sampling/importance_sampling_ratio/min": 0.31646302342414856, "sampling/sampling_logp_difference/max": 1.1505489349365234, "sampling/sampling_logp_difference/mean": 0.03357382118701935, "step": 545 }, { "clip_ratio/high_max": 0.011283236730378121, "clip_ratio/high_mean": 0.011283236730378121, "clip_ratio/low_mean": 0.015035076532512903, "clip_ratio/low_min": 0.015035076532512903, "clip_ratio/region_mean": 0.026318313262891024, "completions/clipped_ratio": 0.0, "completions/max_length": 196.0, "completions/max_terminated_length": 196.0, "completions/mean_length": 168.625, "completions/mean_terminated_length": 168.625, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.3597529251128435, "epoch": 0.02108433734939759, "frac_reward_zero_std": 0.0, "grad_norm": 4.271132469177246, "learning_rate": 8.348484848484849e-06, "loss": 0.0381, "num_tokens": 1173692.0, "reward": 0.9413485527038574, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9413485527038574, "reward_meter_std": 0.09055134654045105, "reward_std": 0.09055135399103165, "reward_total_composite_mean": 0.9413485527038574, "reward_total_composite_std": 0.09055134654045105, "reward_total_mean": 0.9413485527038574, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9413485527038574, "rewards/meter/std": 0.09055134654045105, "rewards/total_composite/mean": 0.9413485527038574, "rewards/total_composite/std": 0.09055134654045105, "sampling/importance_sampling_ratio/max": 1.946368932723999, "sampling/importance_sampling_ratio/mean": 1.0049318075180054, "sampling/importance_sampling_ratio/min": 0.3178619146347046, "sampling/sampling_logp_difference/max": 1.1461381912231445, "sampling/sampling_logp_difference/mean": 0.040303874760866165, "step": 546 }, { "clip_ratio/high_max": 0.032155484426766634, "clip_ratio/high_mean": 0.032155484426766634, "clip_ratio/low_mean": 0.0022522523067891598, "clip_ratio/low_min": 0.0022522523067891598, "clip_ratio/region_mean": 0.034407736733555794, "completions/clipped_ratio": 0.0, "completions/max_length": 127.0, "completions/max_terminated_length": 127.0, "completions/mean_length": 118.125, "completions/mean_terminated_length": 118.125, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.1798835713416338, "epoch": 0.021122953351869013, "frac_reward_zero_std": 0.0, "grad_norm": 3.342482566833496, "learning_rate": 8.345454545454546e-06, "loss": -0.0148, "num_tokens": 1175997.0, "reward": 0.967841625213623, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.967841625213623, "reward_meter_std": 0.06629345566034317, "reward_std": 0.06629344820976257, "reward_total_composite_mean": 0.967841625213623, "reward_total_composite_std": 0.06629345566034317, "reward_total_mean": 0.967841625213623, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.967841625213623, "rewards/meter/std": 0.06629345566034317, "rewards/total_composite/mean": 0.967841625213623, "rewards/total_composite/std": 0.06629345566034317, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0027629137039185, "sampling/importance_sampling_ratio/min": 0.14123374223709106, "sampling/sampling_logp_difference/max": 1.9573390483856201, "sampling/sampling_logp_difference/mean": 0.0339958630502224, "step": 547 }, { "clip_ratio/high_max": 0.025215634261257946, "clip_ratio/high_mean": 0.025215634261257946, "clip_ratio/low_mean": 0.011371727799996734, "clip_ratio/low_min": 0.011371727799996734, "clip_ratio/region_mean": 0.03658736206125468, "completions/clipped_ratio": 0.0, "completions/max_length": 179.0, "completions/max_terminated_length": 179.0, "completions/mean_length": 133.125, "completions/mean_terminated_length": 133.125, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.2751317787915468, "epoch": 0.021161569354340437, "frac_reward_zero_std": 0.0, "grad_norm": 6.153160572052002, "learning_rate": 8.342424242424244e-06, "loss": -0.0406, "num_tokens": 1178566.0, "reward": 0.9537743330001831, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.9785022139549255, "reward_meter_std": 0.02502669021487236, "reward_std": 0.07013029605150223, "reward_total_composite_mean": 0.9537743330001831, "reward_total_composite_std": 0.07013029605150223, "reward_total_mean": 0.9537743330001831, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.9785022139549255, "rewards/meter/std": 0.02502669021487236, "rewards/total_composite/mean": 0.9537743330001831, "rewards/total_composite/std": 0.07013029605150223, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0059552192687988, "sampling/importance_sampling_ratio/min": 0.3159232437610626, "sampling/sampling_logp_difference/max": 1.1522560119628906, "sampling/sampling_logp_difference/mean": 0.035310305655002594, "step": 548 }, { "clip_ratio/high_max": 0.04026081250049174, "clip_ratio/high_mean": 0.04026081250049174, "clip_ratio/low_mean": 0.05864309147000313, "clip_ratio/low_min": 0.05864309147000313, "clip_ratio/region_mean": 0.09890390397049487, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 29.25, "completions/mean_terminated_length": 29.25, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.9557730779051781, "epoch": 0.02120018535681186, "frac_reward_zero_std": 0.0, "grad_norm": 17.468664169311523, "learning_rate": 8.339393939393941e-06, "loss": -0.077, "num_tokens": 1180024.0, "reward": 0.989723801612854, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.989723801612854, "reward_meter_std": 0.011034172028303146, "reward_std": 0.011034182272851467, "reward_total_composite_mean": 0.989723801612854, "reward_total_composite_std": 0.011034172028303146, "reward_total_mean": 0.989723801612854, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.989723801612854, "rewards/meter/std": 0.011034172028303146, "rewards/total_composite/mean": 0.989723801612854, "rewards/total_composite/std": 0.011034172028303146, "sampling/importance_sampling_ratio/max": 1.7388663291931152, "sampling/importance_sampling_ratio/mean": 1.0076111555099487, "sampling/importance_sampling_ratio/min": 0.2602878212928772, "sampling/sampling_logp_difference/max": 1.3459672927856445, "sampling/sampling_logp_difference/mean": 0.10392901301383972, "step": 549 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.021238801359283285, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 8.336363636363636e-06, "loss": 0.0, "num_tokens": 1181720.0, "reward": 0.9139711856842041, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9191176891326904, "reward_count_adherence_std": 0.05388972535729408, "reward_meter_mean": 0.9942543506622314, "reward_meter_std": 0.005934838205575943, "reward_std": 0.056208569556474686, "reward_total_composite_mean": 0.9139711856842041, "reward_total_composite_std": 0.05620856583118439, "reward_total_mean": 0.9139711856842041, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9191176891326904, "rewards/count_adherence/std": 0.05388972535729408, "rewards/meter/mean": 0.9942543506622314, "rewards/meter/std": 0.005934838205575943, "rewards/total_composite/mean": 0.9139711856842041, "rewards/total_composite/std": 0.05620856583118439, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 550 }, { "epoch": 0.021238801359283285, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.125, "eval_completions/max_length": 476.38461538461536, "eval_completions/max_terminated_length": 421.0769230769231, "eval_completions/mean_length": 241.0096153846154, "eval_completions/mean_terminated_length": 199.7985393817608, "eval_completions/min_length": 51.69230769230769, "eval_completions/min_terminated_length": 51.69230769230769, "eval_entropy": 0.1650333639520865, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 1181720.0, "eval_reward": 0.7004609107971191, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.8751364029370822, "eval_reward_count_adherence_std": 0.1485061846100367, "eval_reward_meter_mean": 0.7935248292409457, "eval_reward_meter_std": 0.3153245966308392, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7004609107971191, "eval_reward_total_composite_std": 0.3202407302764746, "eval_reward_total_mean": 0.7004609107971191, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.8751364029370822, "eval_rewards/count_adherence/std": 0.1485061846100367, "eval_rewards/meter/mean": 0.7935248292409457, "eval_rewards/meter/std": 0.3153245966308392, "eval_rewards/total_composite/mean": 0.7004609107971191, "eval_rewards/total_composite/std": 0.3202407302764746, "eval_runtime": 88.1689, "eval_samples_per_second": 1.18, "eval_sampling/importance_sampling_ratio/max": 1.3832710614571204, "eval_sampling/importance_sampling_ratio/mean": 1.0041714814993052, "eval_sampling/importance_sampling_ratio/min": 0.41424351701369655, "eval_sampling/sampling_logp_difference/max": 0.8992016132061298, "eval_sampling/sampling_logp_difference/mean": 0.014547023384903487, "eval_steps_per_second": 0.147, "step": 550 }, { "clip_ratio/high_max": 0.02020994306076318, "clip_ratio/high_mean": 0.02020994306076318, "clip_ratio/low_mean": 0.004437869880348444, "clip_ratio/low_min": 0.004437869880348444, "clip_ratio/region_mean": 0.024647812941111624, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 217.0, "completions/mean_length": 234.375, "completions/mean_terminated_length": 194.71429443359375, "completions/min_length": 169.0, "completions/min_terminated_length": 169.0, "entropy": 0.2244832031428814, "epoch": 0.021277417361754713, "frac_reward_zero_std": 0.0, "grad_norm": 1.8348972797393799, "learning_rate": 8.333333333333334e-06, "loss": -0.1884, "num_tokens": 1184507.0, "reward": 0.637830376625061, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_meter_mean": 0.659835934638977, "reward_meter_std": 0.40453797578811646, "reward_std": 0.39101067185401917, "reward_total_composite_mean": 0.637830376625061, "reward_total_composite_std": 0.39101067185401917, "reward_total_mean": 0.637830376625061, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/meter/mean": 0.659835934638977, "rewards/meter/std": 0.40453797578811646, "rewards/total_composite/mean": 0.637830376625061, "rewards/total_composite/std": 0.39101067185401917, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002615213394165, "sampling/importance_sampling_ratio/min": 0.31355801224708557, "sampling/sampling_logp_difference/max": 1.1597709655761719, "sampling/sampling_logp_difference/mean": 0.03236944600939751, "step": 551 }, { "clip_ratio/high_max": 0.027031495235860348, "clip_ratio/high_mean": 0.027031495235860348, "clip_ratio/low_mean": 0.012516135815531015, "clip_ratio/low_min": 0.012516135815531015, "clip_ratio/region_mean": 0.03954763105139136, "completions/clipped_ratio": 0.0, "completions/max_length": 164.0, "completions/max_terminated_length": 164.0, "completions/mean_length": 147.75, "completions/mean_terminated_length": 147.75, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.31488965079188347, "epoch": 0.021316033364226137, "frac_reward_zero_std": 0.0, "grad_norm": 2.554471731185913, "learning_rate": 8.330303030303031e-06, "loss": 0.015, "num_tokens": 1187017.0, "reward": 0.7965080738067627, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956350326538086, "reward_meter_std": 0.0016549181891605258, "reward_std": 0.0013239351101219654, "reward_total_composite_mean": 0.7965080738067627, "reward_total_composite_std": 0.0013239418622106314, "reward_total_mean": 0.7965080738067627, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956350326538086, "rewards/meter/std": 0.0016549181891605258, "rewards/total_composite/mean": 0.7965080738067627, "rewards/total_composite/std": 0.0013239418622106314, "sampling/importance_sampling_ratio/max": 1.891858458518982, "sampling/importance_sampling_ratio/mean": 1.0011682510375977, "sampling/importance_sampling_ratio/min": 0.1732589304447174, "sampling/sampling_logp_difference/max": 1.7529680728912354, "sampling/sampling_logp_difference/mean": 0.037060461938381195, "step": 552 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.003231151611544192, "clip_ratio/low_min": 0.003231151611544192, "clip_ratio/region_mean": 0.003231151611544192, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 509.75, "completions/mean_terminated_length": 506.0, "completions/min_length": 497.0, "completions/min_terminated_length": 497.0, "entropy": 0.021786156576126814, "epoch": 0.02135464936669756, "frac_reward_zero_std": 0.0, "grad_norm": 0.24584703147411346, "learning_rate": 8.327272727272728e-06, "loss": 0.1244, "num_tokens": 1190519.0, "reward": 0.767181932926178, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7708333730697632, "reward_count_adherence_std": 0.058925554156303406, "reward_meter_mean": 0.9953852891921997, "reward_meter_std": 0.00345088099129498, "reward_std": 0.05727332457900047, "reward_total_composite_mean": 0.767181932926178, "reward_total_composite_std": 0.05727330222725868, "reward_total_mean": 0.767181932926178, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7708333730697632, "rewards/count_adherence/std": 0.058925554156303406, "rewards/meter/mean": 0.9953852891921997, "rewards/meter/std": 0.00345088099129498, "rewards/total_composite/mean": 0.767181932926178, "rewards/total_composite/std": 0.05727330222725868, "sampling/importance_sampling_ratio/max": 1.871092677116394, "sampling/importance_sampling_ratio/mean": 1.0002206563949585, "sampling/importance_sampling_ratio/min": 0.30614548921585083, "sampling/sampling_logp_difference/max": 1.183694839477539, "sampling/sampling_logp_difference/mean": 0.008550068363547325, "step": 553 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0089119115145877, "clip_ratio/low_min": 0.0089119115145877, "clip_ratio/region_mean": 0.0089119115145877, "completions/clipped_ratio": 0.75, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 504.625, "completions/mean_terminated_length": 482.5, "completions/min_length": 468.0, "completions/min_terminated_length": 468.0, "entropy": 0.19928418099880219, "epoch": 0.021393265369168985, "frac_reward_zero_std": 0.0, "grad_norm": 1.8809558153152466, "learning_rate": 8.324242424242425e-06, "loss": 0.378, "num_tokens": 1193244.0, "reward": 0.6255342364311218, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1075671836733818, "reward_meter_mean": 0.7643306255340576, "reward_meter_std": 0.37500372529029846, "reward_std": 0.32354021072387695, "reward_total_composite_mean": 0.6255342364311218, "reward_total_composite_std": 0.32354018092155457, "reward_total_mean": 0.6255342364311218, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1075671836733818, "rewards/meter/mean": 0.7643306255340576, "rewards/meter/std": 0.37500372529029846, "rewards/total_composite/mean": 0.6255342364311218, "rewards/total_composite/std": 0.32354018092155457, "sampling/importance_sampling_ratio/max": 1.7761412858963013, "sampling/importance_sampling_ratio/mean": 1.0125058889389038, "sampling/importance_sampling_ratio/min": 0.27125003933906555, "sampling/sampling_logp_difference/max": 1.3047142028808594, "sampling/sampling_logp_difference/mean": 0.06437436491250992, "step": 554 }, { "clip_ratio/high_max": 0.005685312673449516, "clip_ratio/high_mean": 0.005685312673449516, "clip_ratio/low_mean": 0.017126905964687467, "clip_ratio/low_min": 0.017126905964687467, "clip_ratio/region_mean": 0.022812218638136983, "completions/clipped_ratio": 0.0, "completions/max_length": 136.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 126.375, "completions/mean_terminated_length": 126.375, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.2988105323165655, "epoch": 0.02143188137164041, "frac_reward_zero_std": 0.0, "grad_norm": 2.518230676651001, "learning_rate": 8.321212121212123e-06, "loss": -0.0045, "num_tokens": 1195591.0, "reward": 0.839267909526825, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_meter_mean": 0.9948133826255798, "reward_meter_std": 0.0038075658958405256, "reward_std": 0.12791673839092255, "reward_total_composite_mean": 0.839267909526825, "reward_total_composite_std": 0.12791672348976135, "reward_total_mean": 0.839267909526825, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/meter/mean": 0.9948133826255798, "rewards/meter/std": 0.0038075658958405256, "rewards/total_composite/mean": 0.839267909526825, "rewards/total_composite/std": 0.12791672348976135, "sampling/importance_sampling_ratio/max": 1.7961597442626953, "sampling/importance_sampling_ratio/mean": 1.0079847574234009, "sampling/importance_sampling_ratio/min": 0.2756645977497101, "sampling/sampling_logp_difference/max": 1.2885704040527344, "sampling/sampling_logp_difference/mean": 0.0339997336268425, "step": 555 }, { "clip_ratio/high_max": 0.007534351840149611, "clip_ratio/high_mean": 0.007534351840149611, "clip_ratio/low_mean": 0.0012622967187780887, "clip_ratio/low_min": 0.0012622967187780887, "clip_ratio/region_mean": 0.0087966485589277, "completions/clipped_ratio": 0.0, "completions/max_length": 500.0, "completions/max_terminated_length": 500.0, "completions/mean_length": 482.625, "completions/mean_terminated_length": 482.625, "completions/min_length": 461.0, "completions/min_terminated_length": 461.0, "entropy": 0.06857710564509034, "epoch": 0.021470497374111833, "frac_reward_zero_std": 0.0, "grad_norm": 1.6413999795913696, "learning_rate": 8.318181818181818e-06, "loss": 0.0222, "num_tokens": 1200948.0, "reward": 0.8098647594451904, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.0589255690574646, "reward_meter_mean": 0.9967899918556213, "reward_meter_std": 0.0011801483342424035, "reward_std": 0.058309562504291534, "reward_total_composite_mean": 0.8098647594451904, "reward_total_composite_std": 0.058309562504291534, "reward_total_mean": 0.8098647594451904, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.0589255690574646, "rewards/meter/mean": 0.9967899918556213, "rewards/meter/std": 0.0011801483342424035, "rewards/total_composite/mean": 0.8098647594451904, "rewards/total_composite/std": 0.058309562504291534, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0008271932601929, "sampling/importance_sampling_ratio/min": 0.20025621354579926, "sampling/sampling_logp_difference/max": 1.6081576347351074, "sampling/sampling_logp_difference/mean": 0.010928530246019363, "step": 556 }, { "clip_ratio/high_max": 0.024866676423698664, "clip_ratio/high_mean": 0.024866676423698664, "clip_ratio/low_mean": 0.026261302642524242, "clip_ratio/low_min": 0.026261302642524242, "clip_ratio/region_mean": 0.051127979066222906, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 109.125, "completions/mean_terminated_length": 109.125, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.2295174626633525, "epoch": 0.021509113376583257, "frac_reward_zero_std": 0.0, "grad_norm": 12.06613826751709, "learning_rate": 8.315151515151516e-06, "loss": 0.0447, "num_tokens": 1203421.0, "reward": 0.5651826858520508, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.5895034074783325, "reward_meter_std": 0.44556495547294617, "reward_std": 0.42655691504478455, "reward_total_composite_mean": 0.5651826858520508, "reward_total_composite_std": 0.42655694484710693, "reward_total_mean": 0.5651826858520508, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.5895034074783325, "rewards/meter/std": 0.44556495547294617, "rewards/total_composite/mean": 0.5651826858520508, "rewards/total_composite/std": 0.42655694484710693, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9983038306236267, "sampling/importance_sampling_ratio/min": 0.13523530960083008, "sampling/sampling_logp_difference/max": 2.000739097595215, "sampling/sampling_logp_difference/mean": 0.057171959429979324, "step": 557 }, { "clip_ratio/high_max": 0.025330669013783336, "clip_ratio/high_mean": 0.025330669013783336, "clip_ratio/low_mean": 0.010358731728047132, "clip_ratio/low_min": 0.010358731728047132, "clip_ratio/region_mean": 0.03568940074183047, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.24670910649001598, "epoch": 0.02154772937905468, "frac_reward_zero_std": 0.0, "grad_norm": 4.484973907470703, "learning_rate": 8.312121212121213e-06, "loss": -0.019, "num_tokens": 1205157.0, "reward": 0.9771539568901062, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9771539568901062, "reward_meter_std": 0.036376286298036575, "reward_std": 0.03637627884745598, "reward_total_composite_mean": 0.9771539568901062, "reward_total_composite_std": 0.036376286298036575, "reward_total_mean": 0.9771539568901062, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9771539568901062, "rewards/meter/std": 0.036376286298036575, "rewards/total_composite/mean": 0.9771539568901062, "rewards/total_composite/std": 0.036376286298036575, "sampling/importance_sampling_ratio/max": 1.4410507678985596, "sampling/importance_sampling_ratio/mean": 0.9910410046577454, "sampling/importance_sampling_ratio/min": 0.17207419872283936, "sampling/sampling_logp_difference/max": 1.7598295211791992, "sampling/sampling_logp_difference/mean": 0.04534757509827614, "step": 558 }, { "clip_ratio/high_max": 0.027243590680882335, "clip_ratio/high_mean": 0.027243590680882335, "clip_ratio/low_mean": 0.013174113817512989, "clip_ratio/low_min": 0.013174113817512989, "clip_ratio/region_mean": 0.040417704498395324, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 55.125, "completions/mean_terminated_length": 55.125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.3850720599293709, "epoch": 0.021586345381526106, "frac_reward_zero_std": 0.0, "grad_norm": 12.47499942779541, "learning_rate": 8.30909090909091e-06, "loss": 0.0143, "num_tokens": 1206934.0, "reward": 0.9845154285430908, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9845154285430908, "reward_meter_std": 0.022365685552358627, "reward_std": 0.022365683689713478, "reward_total_composite_mean": 0.9845154285430908, "reward_total_composite_std": 0.022365685552358627, "reward_total_mean": 0.9845154285430908, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9845154285430908, "rewards/meter/std": 0.022365685552358627, "rewards/total_composite/mean": 0.9845154285430908, "rewards/total_composite/std": 0.022365685552358627, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0180634260177612, "sampling/importance_sampling_ratio/min": 0.3067775070667267, "sampling/sampling_logp_difference/max": 1.1816325187683105, "sampling/sampling_logp_difference/mean": 0.05525405332446098, "step": 559 }, { "clip_ratio/high_max": 0.029417280456982553, "clip_ratio/high_mean": 0.029417280456982553, "clip_ratio/low_mean": 0.004687500186264515, "clip_ratio/low_min": 0.004687500186264515, "clip_ratio/region_mean": 0.03410478064324707, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 86.875, "completions/mean_terminated_length": 86.875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.22090137097984552, "epoch": 0.02162496138399753, "frac_reward_zero_std": 0.0, "grad_norm": 7.248865604400635, "learning_rate": 8.306060606060606e-06, "loss": -0.0201, "num_tokens": 1209005.0, "reward": 0.923915684223175, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.923915684223175, "reward_meter_std": 0.18663685023784637, "reward_std": 0.18663683533668518, "reward_total_composite_mean": 0.923915684223175, "reward_total_composite_std": 0.18663685023784637, "reward_total_mean": 0.923915684223175, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.923915684223175, "rewards/meter/std": 0.18663685023784637, "rewards/total_composite/mean": 0.923915684223175, "rewards/total_composite/std": 0.18663685023784637, "sampling/importance_sampling_ratio/max": 1.8499445915222168, "sampling/importance_sampling_ratio/mean": 1.0030776262283325, "sampling/importance_sampling_ratio/min": 0.3106648325920105, "sampling/sampling_logp_difference/max": 1.1690406799316406, "sampling/sampling_logp_difference/mean": 0.03298182412981987, "step": 560 }, { "clip_ratio/high_max": 0.04831800376996398, "clip_ratio/high_mean": 0.04831800376996398, "clip_ratio/low_mean": 0.005494505632668734, "clip_ratio/low_min": 0.005494505632668734, "clip_ratio/region_mean": 0.05381250940263271, "completions/clipped_ratio": 0.0, "completions/max_length": 110.0, "completions/max_terminated_length": 110.0, "completions/mean_length": 96.875, "completions/mean_terminated_length": 96.875, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.5123469866812229, "epoch": 0.021663577386468954, "frac_reward_zero_std": 0.0, "grad_norm": 5.603593826293945, "learning_rate": 8.303030303030305e-06, "loss": -0.009, "num_tokens": 1211172.0, "reward": 0.9536159038543701, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.995019793510437, "reward_meter_std": 0.0025853195693343878, "reward_std": 0.11767084896564484, "reward_total_composite_mean": 0.9536159038543701, "reward_total_composite_std": 0.11767086386680603, "reward_total_mean": 0.9536159038543701, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.995019793510437, "rewards/meter/std": 0.0025853195693343878, "rewards/total_composite/mean": 0.9536159038543701, "rewards/total_composite/std": 0.11767086386680603, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0120702981948853, "sampling/importance_sampling_ratio/min": 0.07194232940673828, "sampling/sampling_logp_difference/max": 2.6318905353546143, "sampling/sampling_logp_difference/mean": 0.05422169715166092, "step": 561 }, { "clip_ratio/high_max": 0.06236687349155545, "clip_ratio/high_mean": 0.06236687349155545, "clip_ratio/low_mean": 0.0078125, "clip_ratio/low_min": 0.0078125, "clip_ratio/region_mean": 0.07017937349155545, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 37.75, "completions/mean_terminated_length": 37.75, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.7698529586195946, "epoch": 0.021702193388940378, "frac_reward_zero_std": 0.0, "grad_norm": 11.571051597595215, "learning_rate": 8.3e-06, "loss": 0.2898, "num_tokens": 1212666.0, "reward": 0.8193449378013611, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.9433770179748535, "reward_meter_std": 0.13426972925662994, "reward_std": 0.3567107319831848, "reward_total_composite_mean": 0.8193449378013611, "reward_total_composite_std": 0.3567107319831848, "reward_total_mean": 0.8193449378013611, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.9433770179748535, "rewards/meter/std": 0.13426972925662994, "rewards/total_composite/mean": 0.8193449378013611, "rewards/total_composite/std": 0.3567107319831848, "sampling/importance_sampling_ratio/max": 1.9018025398254395, "sampling/importance_sampling_ratio/mean": 1.0174795389175415, "sampling/importance_sampling_ratio/min": 0.22708770632743835, "sampling/sampling_logp_difference/max": 1.4824190139770508, "sampling/sampling_logp_difference/mean": 0.09738308191299438, "step": 562 }, { "clip_ratio/high_max": 0.02509535406716168, "clip_ratio/high_mean": 0.02509535406716168, "clip_ratio/low_mean": 0.008614832302555442, "clip_ratio/low_min": 0.008614832302555442, "clip_ratio/region_mean": 0.03371018636971712, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 100.125, "completions/mean_terminated_length": 100.125, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.4031477700918913, "epoch": 0.021740809391411802, "frac_reward_zero_std": 0.0, "grad_norm": 4.293374061584473, "learning_rate": 8.296969696969697e-06, "loss": 0.0224, "num_tokens": 1214907.0, "reward": 0.7916634678840637, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7916634678840637, "reward_meter_std": 0.35509946942329407, "reward_std": 0.3550994396209717, "reward_total_composite_mean": 0.7916634678840637, "reward_total_composite_std": 0.35509946942329407, "reward_total_mean": 0.7916634678840637, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7916634678840637, "rewards/meter/std": 0.35509946942329407, "rewards/total_composite/mean": 0.7916634678840637, "rewards/total_composite/std": 0.35509946942329407, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0102379322052002, "sampling/importance_sampling_ratio/min": 0.38540852069854736, "sampling/sampling_logp_difference/max": 0.95345139503479, "sampling/sampling_logp_difference/mean": 0.04964831843972206, "step": 563 }, { "clip_ratio/high_max": 0.044469698797911406, "clip_ratio/high_mean": 0.044469698797911406, "clip_ratio/low_mean": 0.031700728461146355, "clip_ratio/low_min": 0.031700728461146355, "clip_ratio/region_mean": 0.07617042725905776, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 59.125, "completions/mean_terminated_length": 59.125, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.574961706995964, "epoch": 0.021779425393883226, "frac_reward_zero_std": 0.0, "grad_norm": 7.8068928718566895, "learning_rate": 8.293939393939395e-06, "loss": -0.1095, "num_tokens": 1216596.0, "reward": 0.9822205305099487, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9822205305099487, "reward_meter_std": 0.0145874610170722, "reward_std": 0.014587470330297947, "reward_total_composite_mean": 0.9822205305099487, "reward_total_composite_std": 0.0145874610170722, "reward_total_mean": 0.9822205305099487, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9822205305099487, "rewards/meter/std": 0.0145874610170722, "rewards/total_composite/mean": 0.9822205305099487, "rewards/total_composite/std": 0.0145874610170722, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0125072002410889, "sampling/importance_sampling_ratio/min": 0.07002489268779755, "sampling/sampling_logp_difference/max": 2.658904552459717, "sampling/sampling_logp_difference/mean": 0.0998057872056961, "step": 564 }, { "clip_ratio/high_max": 0.056833125185221434, "clip_ratio/high_mean": 0.056833125185221434, "clip_ratio/low_mean": 0.021110056899487972, "clip_ratio/low_min": 0.021110056899487972, "clip_ratio/region_mean": 0.0779431820847094, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 36.375, "completions/mean_terminated_length": 36.375, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.8591618165373802, "epoch": 0.02181804139635465, "frac_reward_zero_std": 0.0, "grad_norm": 9.879420280456543, "learning_rate": 8.290909090909092e-06, "loss": 0.1893, "num_tokens": 1218119.0, "reward": 0.7292733192443848, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.7298904061317444, "reward_meter_std": 0.45080265402793884, "reward_std": 0.451938658952713, "reward_total_composite_mean": 0.7292733192443848, "reward_total_composite_std": 0.4519386887550354, "reward_total_mean": 0.7292733192443848, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.7298904061317444, "rewards/meter/std": 0.45080265402793884, "rewards/total_composite/mean": 0.7292733192443848, "rewards/total_composite/std": 0.4519386887550354, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0224130153656006, "sampling/importance_sampling_ratio/min": 0.3525867760181427, "sampling/sampling_logp_difference/max": 1.0424585342407227, "sampling/sampling_logp_difference/mean": 0.09998878836631775, "step": 565 }, { "clip_ratio/high_max": 0.019384657382033765, "clip_ratio/high_mean": 0.019384657382033765, "clip_ratio/low_mean": 0.03550754114985466, "clip_ratio/low_min": 0.03550754114985466, "clip_ratio/region_mean": 0.054892198531888425, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 73.0, "completions/mean_terminated_length": 73.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.5164803229272366, "epoch": 0.021856657398826074, "frac_reward_zero_std": 0.0, "grad_norm": 7.685236930847168, "learning_rate": 8.287878787878787e-06, "loss": -0.0217, "num_tokens": 1220231.0, "reward": 0.8192075490951538, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8192075490951538, "reward_meter_std": 0.1594366729259491, "reward_std": 0.1594366729259491, "reward_total_composite_mean": 0.8192075490951538, "reward_total_composite_std": 0.1594366729259491, "reward_total_mean": 0.8192075490951538, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8192075490951538, "rewards/meter/std": 0.1594366729259491, "rewards/total_composite/mean": 0.8192075490951538, "rewards/total_composite/std": 0.1594366729259491, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.019374132156372, "sampling/importance_sampling_ratio/min": 0.24237361550331116, "sampling/sampling_logp_difference/max": 1.4172749519348145, "sampling/sampling_logp_difference/mean": 0.07931912690401077, "step": 566 }, { "clip_ratio/high_max": 0.05356209189631045, "clip_ratio/high_mean": 0.05356209189631045, "clip_ratio/low_mean": 0.06415719725191593, "clip_ratio/low_min": 0.06415719725191593, "clip_ratio/region_mean": 0.11771928914822638, "completions/clipped_ratio": 0.0, "completions/max_length": 51.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 48.25, "completions/mean_terminated_length": 48.25, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.7732437737286091, "epoch": 0.0218952734012975, "frac_reward_zero_std": 0.0, "grad_norm": 13.685770988464355, "learning_rate": 8.284848484848486e-06, "loss": 0.0125, "num_tokens": 1221897.0, "reward": 0.7687733173370361, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7687733173370361, "reward_meter_std": 0.23431365191936493, "reward_std": 0.23431365191936493, "reward_total_composite_mean": 0.7687733173370361, "reward_total_composite_std": 0.23431365191936493, "reward_total_mean": 0.7687733173370361, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7687733173370361, "rewards/meter/std": 0.23431365191936493, "rewards/total_composite/mean": 0.7687733173370361, "rewards/total_composite/std": 0.23431365191936493, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0146090984344482, "sampling/importance_sampling_ratio/min": 0.15560193359851837, "sampling/sampling_logp_difference/max": 1.8604542016983032, "sampling/sampling_logp_difference/mean": 0.12316906452178955, "step": 567 }, { "clip_ratio/high_max": 0.07650120463222265, "clip_ratio/high_mean": 0.07650120463222265, "clip_ratio/low_mean": 0.07042828761041164, "clip_ratio/low_min": 0.07042828761041164, "clip_ratio/region_mean": 0.1469294922426343, "completions/clipped_ratio": 0.0, "completions/max_length": 51.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 41.625, "completions/mean_terminated_length": 41.625, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 1.1647358424961567, "epoch": 0.021933889403768923, "frac_reward_zero_std": 0.0, "grad_norm": 12.740221977233887, "learning_rate": 8.281818181818182e-06, "loss": 0.0232, "num_tokens": 1223686.0, "reward": 0.3464130163192749, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3464130163192749, "reward_meter_std": 0.327831894159317, "reward_std": 0.327831894159317, "reward_total_composite_mean": 0.3464130163192749, "reward_total_composite_std": 0.327831894159317, "reward_total_mean": 0.3464130163192749, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3464130163192749, "rewards/meter/std": 0.327831894159317, "rewards/total_composite/mean": 0.3464130163192749, "rewards/total_composite/std": 0.327831894159317, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0059748888015747, "sampling/importance_sampling_ratio/min": 0.15346065163612366, "sampling/sampling_logp_difference/max": 1.874311089515686, "sampling/sampling_logp_difference/mean": 0.13335032761096954, "step": 568 }, { "clip_ratio/high_max": 0.020774202537722886, "clip_ratio/high_mean": 0.020774202537722886, "clip_ratio/low_mean": 0.0090304184705019, "clip_ratio/low_min": 0.0090304184705019, "clip_ratio/region_mean": 0.029804621008224785, "completions/clipped_ratio": 0.0, "completions/max_length": 263.0, "completions/max_terminated_length": 263.0, "completions/mean_length": 208.0, "completions/mean_terminated_length": 208.0, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.30806298181414604, "epoch": 0.021972505406240347, "frac_reward_zero_std": 0.0, "grad_norm": 5.267251014709473, "learning_rate": 8.27878787878788e-06, "loss": 0.0953, "num_tokens": 1226918.0, "reward": 0.8531581163406372, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9642857313156128, "reward_count_adherence_std": 0.06613000482320786, "reward_meter_mean": 0.8887746334075928, "reward_meter_std": 0.28811898827552795, "reward_std": 0.28023195266723633, "reward_total_composite_mean": 0.8531581163406372, "reward_total_composite_std": 0.28023195266723633, "reward_total_mean": 0.8531581163406372, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9642857313156128, "rewards/count_adherence/std": 0.06613000482320786, "rewards/meter/mean": 0.8887746334075928, "rewards/meter/std": 0.28811898827552795, "rewards/total_composite/mean": 0.8531581163406372, "rewards/total_composite/std": 0.28023195266723633, "sampling/importance_sampling_ratio/max": 1.8939993381500244, "sampling/importance_sampling_ratio/mean": 1.004589319229126, "sampling/importance_sampling_ratio/min": 0.07200159877538681, "sampling/sampling_logp_difference/max": 2.6310670375823975, "sampling/sampling_logp_difference/mean": 0.04299961030483246, "step": 569 }, { "clip_ratio/high_max": 0.04666911787353456, "clip_ratio/high_mean": 0.04666911787353456, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/region_mean": 0.05235093622468412, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.3545149974524975, "epoch": 0.02201112140871177, "frac_reward_zero_std": 0.0, "grad_norm": 4.654249668121338, "learning_rate": 8.275757575757577e-06, "loss": 0.0113, "num_tokens": 1228806.0, "reward": 0.984466016292572, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.984466016292572, "reward_meter_std": 0.027155917137861252, "reward_std": 0.027155913412570953, "reward_total_composite_mean": 0.984466016292572, "reward_total_composite_std": 0.027155917137861252, "reward_total_mean": 0.984466016292572, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.984466016292572, "rewards/meter/std": 0.027155917137861252, "rewards/total_composite/mean": 0.984466016292572, "rewards/total_composite/std": 0.027155917137861252, "sampling/importance_sampling_ratio/max": 1.9392898082733154, "sampling/importance_sampling_ratio/mean": 1.0072888135910034, "sampling/importance_sampling_ratio/min": 0.3174997866153717, "sampling/sampling_logp_difference/max": 1.147278070449829, "sampling/sampling_logp_difference/mean": 0.053486477583646774, "step": 570 }, { "clip_ratio/high_max": 0.014448553905822337, "clip_ratio/high_mean": 0.014448553905822337, "clip_ratio/low_mean": 0.015158275258727372, "clip_ratio/low_min": 0.015158275258727372, "clip_ratio/region_mean": 0.02960682916454971, "completions/clipped_ratio": 0.0, "completions/max_length": 419.0, "completions/max_terminated_length": 419.0, "completions/mean_length": 372.375, "completions/mean_terminated_length": 372.375, "completions/min_length": 329.0, "completions/min_terminated_length": 329.0, "entropy": 0.16989378002472222, "epoch": 0.022049737411183195, "frac_reward_zero_std": 0.0, "grad_norm": 2.5224688053131104, "learning_rate": 8.272727272727274e-06, "loss": -0.01, "num_tokens": 1233705.0, "reward": 0.3962605595588684, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.09234060347080231, "reward_meter_mean": 0.5269562005996704, "reward_meter_std": 0.4027676284313202, "reward_std": 0.30246567726135254, "reward_total_composite_mean": 0.3962605595588684, "reward_total_composite_std": 0.30246567726135254, "reward_total_mean": 0.3962605595588684, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.09234060347080231, "rewards/meter/mean": 0.5269562005996704, "rewards/meter/std": 0.4027676284313202, "rewards/total_composite/mean": 0.3962605595588684, "rewards/total_composite/std": 0.30246567726135254, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9997126460075378, "sampling/importance_sampling_ratio/min": 0.11277057975530624, "sampling/sampling_logp_difference/max": 2.1823997497558594, "sampling/sampling_logp_difference/mean": 0.0314236655831337, "step": 571 }, { "clip_ratio/high_max": 0.04550157627090812, "clip_ratio/high_mean": 0.04550157627090812, "clip_ratio/low_mean": 0.025765900732949376, "clip_ratio/low_min": 0.025765900732949376, "clip_ratio/region_mean": 0.0712674770038575, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 79.25, "completions/mean_terminated_length": 79.25, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.6656168103218079, "epoch": 0.02208835341365462, "frac_reward_zero_std": 0.0, "grad_norm": 8.577203750610352, "learning_rate": 8.269696969696971e-06, "loss": -0.0052, "num_tokens": 1235635.0, "reward": 0.6216261386871338, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6216261386871338, "reward_meter_std": 0.42476242780685425, "reward_std": 0.42476242780685425, "reward_total_composite_mean": 0.6216261386871338, "reward_total_composite_std": 0.42476242780685425, "reward_total_mean": 0.6216261386871338, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6216261386871338, "rewards/meter/std": 0.42476242780685425, "rewards/total_composite/mean": 0.6216261386871338, "rewards/total_composite/std": 0.42476242780685425, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0138676166534424, "sampling/importance_sampling_ratio/min": 0.27314189076423645, "sampling/sampling_logp_difference/max": 1.9287548065185547, "sampling/sampling_logp_difference/mean": 0.08498334139585495, "step": 572 }, { "clip_ratio/high_max": 0.08289422187954187, "clip_ratio/high_mean": 0.08289422187954187, "clip_ratio/low_mean": 0.023148147389292717, "clip_ratio/low_min": 0.023148147389292717, "clip_ratio/region_mean": 0.10604236926883459, "completions/clipped_ratio": 0.0, "completions/max_length": 31.0, "completions/max_terminated_length": 31.0, "completions/mean_length": 28.875, "completions/mean_terminated_length": 28.875, "completions/min_length": 25.0, "completions/min_terminated_length": 25.0, "entropy": 1.1661205813288689, "epoch": 0.022126969416126043, "frac_reward_zero_std": 0.0, "grad_norm": 14.97447681427002, "learning_rate": 8.266666666666667e-06, "loss": -0.0174, "num_tokens": 1237082.0, "reward": 0.8920953273773193, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8920953273773193, "reward_meter_std": 0.2680381238460541, "reward_std": 0.2680380940437317, "reward_total_composite_mean": 0.8920953273773193, "reward_total_composite_std": 0.2680381238460541, "reward_total_mean": 0.8920953273773193, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8920953273773193, "rewards/meter/std": 0.2680381238460541, "rewards/total_composite/mean": 0.8920953273773193, "rewards/total_composite/std": 0.2680381238460541, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0461758375167847, "sampling/importance_sampling_ratio/min": 0.3460072875022888, "sampling/sampling_logp_difference/max": 1.061295509338379, "sampling/sampling_logp_difference/mean": 0.12218277901411057, "step": 573 }, { "clip_ratio/high_max": 0.011733462568372488, "clip_ratio/high_mean": 0.011733462568372488, "clip_ratio/low_mean": 0.01109237689524889, "clip_ratio/low_min": 0.01109237689524889, "clip_ratio/region_mean": 0.022825839463621378, "completions/clipped_ratio": 0.0, "completions/max_length": 265.0, "completions/max_terminated_length": 265.0, "completions/mean_length": 242.375, "completions/mean_terminated_length": 242.375, "completions/min_length": 208.0, "completions/min_terminated_length": 208.0, "entropy": 0.18924708105623722, "epoch": 0.022165585418597467, "frac_reward_zero_std": 0.0, "grad_norm": 1.9361296892166138, "learning_rate": 8.263636363636366e-06, "loss": 0.0108, "num_tokens": 1240781.0, "reward": 0.7266948223114014, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.930555522441864, "reward_count_adherence_std": 0.05750546231865883, "reward_meter_mean": 0.7801742553710938, "reward_meter_std": 0.3070351183414459, "reward_std": 0.2840970754623413, "reward_total_composite_mean": 0.7266948223114014, "reward_total_composite_std": 0.2840970754623413, "reward_total_mean": 0.7266948223114014, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.930555522441864, "rewards/count_adherence/std": 0.05750546231865883, "rewards/meter/mean": 0.7801742553710938, "rewards/meter/std": 0.3070351183414459, "rewards/total_composite/mean": 0.7266948223114014, "rewards/total_composite/std": 0.2840970754623413, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0045069456100464, "sampling/importance_sampling_ratio/min": 0.050895627588033676, "sampling/sampling_logp_difference/max": 2.977978229522705, "sampling/sampling_logp_difference/mean": 0.03169988840818405, "step": 574 }, { "clip_ratio/high_max": 0.05042124609462917, "clip_ratio/high_mean": 0.05042124609462917, "clip_ratio/low_mean": 0.017729461658746004, "clip_ratio/low_min": 0.017729461658746004, "clip_ratio/region_mean": 0.06815070775337517, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 90.0, "completions/mean_terminated_length": 90.0, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.570578096434474, "epoch": 0.02220420142106889, "frac_reward_zero_std": 0.0, "grad_norm": 6.091910362243652, "learning_rate": 8.260606060606061e-06, "loss": -0.0409, "num_tokens": 1242957.0, "reward": 0.934752345085144, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.934752345085144, "reward_meter_std": 0.10161388665437698, "reward_std": 0.10161387175321579, "reward_total_composite_mean": 0.934752345085144, "reward_total_composite_std": 0.10161388665437698, "reward_total_mean": 0.934752345085144, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.934752345085144, "rewards/meter/std": 0.10161388665437698, "rewards/total_composite/mean": 0.934752345085144, "rewards/total_composite/std": 0.10161388665437698, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.009366750717163, "sampling/importance_sampling_ratio/min": 0.2479490041732788, "sampling/sampling_logp_difference/max": 1.3945322036743164, "sampling/sampling_logp_difference/mean": 0.07466892898082733, "step": 575 }, { "clip_ratio/high_max": 0.08125000167638063, "clip_ratio/high_mean": 0.08125000167638063, "clip_ratio/low_mean": 0.03163566067814827, "clip_ratio/low_min": 0.03163566067814827, "clip_ratio/region_mean": 0.1128856623545289, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 32.375, "completions/mean_terminated_length": 32.375, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 1.167568787932396, "epoch": 0.022242817423540315, "frac_reward_zero_std": 0.0, "grad_norm": 18.79474639892578, "learning_rate": 8.257575757575758e-06, "loss": 0.0045, "num_tokens": 1244488.0, "reward": 0.6495904922485352, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6495904922485352, "reward_meter_std": 0.46892398595809937, "reward_std": 0.46892398595809937, "reward_total_composite_mean": 0.6495904922485352, "reward_total_composite_std": 0.46892398595809937, "reward_total_mean": 0.6495904922485352, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6495904922485352, "rewards/meter/std": 0.46892398595809937, "rewards/total_composite/mean": 0.6495904922485352, "rewards/total_composite/std": 0.46892398595809937, "sampling/importance_sampling_ratio/max": 1.6851778030395508, "sampling/importance_sampling_ratio/mean": 1.0079195499420166, "sampling/importance_sampling_ratio/min": 0.3687765896320343, "sampling/sampling_logp_difference/max": 0.9975643157958984, "sampling/sampling_logp_difference/mean": 0.11486772447824478, "step": 576 }, { "clip_ratio/high_max": 0.02569850441068411, "clip_ratio/high_mean": 0.02569850441068411, "clip_ratio/low_mean": 0.030232772696763277, "clip_ratio/low_min": 0.030232772696763277, "clip_ratio/region_mean": 0.055931277107447386, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 94.625, "completions/mean_terminated_length": 94.625, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.5516252182424068, "epoch": 0.02228143342601174, "frac_reward_zero_std": 0.0, "grad_norm": 4.967631816864014, "learning_rate": 8.254545454545456e-06, "loss": -0.0136, "num_tokens": 1246653.0, "reward": 0.5170981884002686, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9166666865348816, "reward_count_adherence_std": 0.15430334210395813, "reward_meter_mean": 0.5609345436096191, "reward_meter_std": 0.46121135354042053, "reward_std": 0.43609705567359924, "reward_total_composite_mean": 0.5170981884002686, "reward_total_composite_std": 0.43609705567359924, "reward_total_mean": 0.5170981884002686, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9166666865348816, "rewards/count_adherence/std": 0.15430334210395813, "rewards/meter/mean": 0.5609345436096191, "rewards/meter/std": 0.46121135354042053, "rewards/total_composite/mean": 0.5170981884002686, "rewards/total_composite/std": 0.43609705567359924, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0093859434127808, "sampling/importance_sampling_ratio/min": 0.17441149055957794, "sampling/sampling_logp_difference/max": 1.746337890625, "sampling/sampling_logp_difference/mean": 0.07433033734560013, "step": 577 }, { "clip_ratio/high_max": 0.02754599740728736, "clip_ratio/high_mean": 0.02754599740728736, "clip_ratio/low_mean": 0.024840134428814054, "clip_ratio/low_min": 0.024840134428814054, "clip_ratio/region_mean": 0.05238613183610141, "completions/clipped_ratio": 0.0, "completions/max_length": 188.0, "completions/max_terminated_length": 188.0, "completions/mean_length": 164.125, "completions/mean_terminated_length": 164.125, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.5197625793516636, "epoch": 0.022320049428483164, "frac_reward_zero_std": 0.0, "grad_norm": 3.93625545501709, "learning_rate": 8.251515151515153e-06, "loss": 0.0284, "num_tokens": 1249486.0, "reward": 0.9166064262390137, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9166064262390137, "reward_meter_std": 0.09252025932073593, "reward_std": 0.09252026677131653, "reward_total_composite_mean": 0.9166064262390137, "reward_total_composite_std": 0.09252025932073593, "reward_total_mean": 0.9166064262390137, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9166064262390137, "rewards/meter/std": 0.09252025932073593, "rewards/total_composite/mean": 0.9166064262390137, "rewards/total_composite/std": 0.09252025932073593, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0109132528305054, "sampling/importance_sampling_ratio/min": 0.29223909974098206, "sampling/sampling_logp_difference/max": 1.4260609149932861, "sampling/sampling_logp_difference/mean": 0.07041816413402557, "step": 578 }, { "clip_ratio/high_max": 0.06810012133792043, "clip_ratio/high_mean": 0.06810012133792043, "clip_ratio/low_mean": 0.01710526365786791, "clip_ratio/low_min": 0.01710526365786791, "clip_ratio/region_mean": 0.08520538499578834, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 69.25, "completions/mean_terminated_length": 69.25, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.8376755490899086, "epoch": 0.022358665430954588, "frac_reward_zero_std": 0.0, "grad_norm": 9.262944221496582, "learning_rate": 8.248484848484848e-06, "loss": 0.1901, "num_tokens": 1251320.0, "reward": 0.9087815284729004, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9710856676101685, "reward_meter_std": 0.049965545535087585, "reward_std": 0.17285725474357605, "reward_total_composite_mean": 0.9087815284729004, "reward_total_composite_std": 0.17285725474357605, "reward_total_mean": 0.9087815284729004, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9710856676101685, "rewards/meter/std": 0.049965545535087585, "rewards/total_composite/mean": 0.9087815284729004, "rewards/total_composite/std": 0.17285725474357605, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0127062797546387, "sampling/importance_sampling_ratio/min": 0.3225191533565521, "sampling/sampling_logp_difference/max": 1.1315927505493164, "sampling/sampling_logp_difference/mean": 0.096940778195858, "step": 579 }, { "clip_ratio/high_max": 0.018500169040635228, "clip_ratio/high_mean": 0.018500169040635228, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.018500169040635228, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 362.625, "completions/mean_terminated_length": 341.2857360839844, "completions/min_length": 316.0, "completions/min_terminated_length": 316.0, "entropy": 0.16224890854209661, "epoch": 0.022397281433426012, "frac_reward_zero_std": 0.0, "grad_norm": 0.5297556519508362, "learning_rate": 8.245454545454546e-06, "loss": -0.2893, "num_tokens": 1255245.0, "reward": 0.4977211654186249, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.5588235855102539, "reward_count_adherence_std": 0.04446641355752945, "reward_meter_mean": 0.9441625475883484, "reward_meter_std": 0.14516927301883698, "reward_std": 0.2027885913848877, "reward_total_composite_mean": 0.4977211654186249, "reward_total_composite_std": 0.2027886062860489, "reward_total_mean": 0.4977211654186249, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.5588235855102539, "rewards/count_adherence/std": 0.04446641355752945, "rewards/meter/mean": 0.9441625475883484, "rewards/meter/std": 0.14516927301883698, "rewards/total_composite/mean": 0.4977211654186249, "rewards/total_composite/std": 0.2027886062860489, "sampling/importance_sampling_ratio/max": 1.8625940084457397, "sampling/importance_sampling_ratio/mean": 1.0028082132339478, "sampling/importance_sampling_ratio/min": 0.29588600993156433, "sampling/sampling_logp_difference/max": 1.2177810668945312, "sampling/sampling_logp_difference/mean": 0.02394985593855381, "step": 580 }, { "clip_ratio/high_max": 0.07077483460307121, "clip_ratio/high_mean": 0.07077483460307121, "clip_ratio/low_mean": 0.02420289907604456, "clip_ratio/low_min": 0.02420289907604456, "clip_ratio/region_mean": 0.09497773367911577, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 67.875, "completions/mean_terminated_length": 67.875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.8226092867553234, "epoch": 0.022435897435897436, "frac_reward_zero_std": 0.0, "grad_norm": 6.394820690155029, "learning_rate": 8.242424242424243e-06, "loss": 0.0445, "num_tokens": 1257068.0, "reward": 0.9752446413040161, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9752446413040161, "reward_meter_std": 0.03254934400320053, "reward_std": 0.03254932910203934, "reward_total_composite_mean": 0.9752446413040161, "reward_total_composite_std": 0.03254934400320053, "reward_total_mean": 0.9752446413040161, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9752446413040161, "rewards/meter/std": 0.03254934400320053, "rewards/total_composite/mean": 0.9752446413040161, "rewards/total_composite/std": 0.03254934400320053, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0144362449645996, "sampling/importance_sampling_ratio/min": 0.20634615421295166, "sampling/sampling_logp_difference/max": 1.578200101852417, "sampling/sampling_logp_difference/mean": 0.09171347320079803, "step": 581 }, { "clip_ratio/high_max": 0.010697707650251687, "clip_ratio/high_mean": 0.010697707650251687, "clip_ratio/low_mean": 0.036390340072102845, "clip_ratio/low_min": 0.036390340072102845, "clip_ratio/region_mean": 0.04708804772235453, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 59.625, "completions/mean_terminated_length": 59.625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.32135884277522564, "epoch": 0.02247451343836886, "frac_reward_zero_std": 0.0, "grad_norm": 3.4065492153167725, "learning_rate": 8.23939393939394e-06, "loss": -0.0091, "num_tokens": 1258849.0, "reward": 0.9942880868911743, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942880868911743, "reward_meter_std": 0.002081402810290456, "reward_std": 0.0020814072340726852, "reward_total_composite_mean": 0.9942880868911743, "reward_total_composite_std": 0.002081402810290456, "reward_total_mean": 0.9942880868911743, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942880868911743, "rewards/meter/std": 0.002081402810290456, "rewards/total_composite/mean": 0.9942880868911743, "rewards/total_composite/std": 0.002081402810290456, "sampling/importance_sampling_ratio/max": 1.7408068180084229, "sampling/importance_sampling_ratio/mean": 1.0069268941879272, "sampling/importance_sampling_ratio/min": 0.3066324293613434, "sampling/sampling_logp_difference/max": 1.182105541229248, "sampling/sampling_logp_difference/mean": 0.04282836988568306, "step": 582 }, { "clip_ratio/high_max": 0.02780228859046474, "clip_ratio/high_mean": 0.02780228859046474, "clip_ratio/low_mean": 0.02640543133020401, "clip_ratio/low_min": 0.02640543133020401, "clip_ratio/region_mean": 0.05420771992066875, "completions/clipped_ratio": 0.0, "completions/max_length": 166.0, "completions/max_terminated_length": 166.0, "completions/mean_length": 147.125, "completions/mean_terminated_length": 147.125, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.4634787142276764, "epoch": 0.022513129440840284, "frac_reward_zero_std": 0.0, "grad_norm": 5.01640510559082, "learning_rate": 8.236363636363637e-06, "loss": 0.0741, "num_tokens": 1261386.0, "reward": 0.6761964559555054, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.0890870913863182, "reward_meter_mean": 0.7399221062660217, "reward_meter_std": 0.3321729898452759, "reward_std": 0.29238972067832947, "reward_total_composite_mean": 0.6761964559555054, "reward_total_composite_std": 0.29238972067832947, "reward_total_mean": 0.6761964559555054, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.0890870913863182, "rewards/meter/mean": 0.7399221062660217, "rewards/meter/std": 0.3321729898452759, "rewards/total_composite/mean": 0.6761964559555054, "rewards/total_composite/std": 0.29238972067832947, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0108050107955933, "sampling/importance_sampling_ratio/min": 0.16547901928424835, "sampling/sampling_logp_difference/max": 1.7989108562469482, "sampling/sampling_logp_difference/mean": 0.06298165768384933, "step": 583 }, { "clip_ratio/high_max": 0.05211255559697747, "clip_ratio/high_mean": 0.05211255559697747, "clip_ratio/low_mean": 0.02927643246948719, "clip_ratio/low_min": 0.02927643246948719, "clip_ratio/region_mean": 0.08138898806646466, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 1.1980116702616215, "epoch": 0.02255174544331171, "frac_reward_zero_std": 0.0, "grad_norm": 8.98315715789795, "learning_rate": 8.233333333333335e-06, "loss": 0.0272, "num_tokens": 1263178.0, "reward": 0.7797154188156128, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7797154188156128, "reward_meter_std": 0.2954230308532715, "reward_std": 0.29542306065559387, "reward_total_composite_mean": 0.7797154188156128, "reward_total_composite_std": 0.2954230308532715, "reward_total_mean": 0.7797154188156128, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7797154188156128, "rewards/meter/std": 0.2954230308532715, "rewards/total_composite/mean": 0.7797154188156128, "rewards/total_composite/std": 0.2954230308532715, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0329049825668335, "sampling/importance_sampling_ratio/min": 0.3161243200302124, "sampling/sampling_logp_difference/max": 1.1516196727752686, "sampling/sampling_logp_difference/mean": 0.10191215574741364, "step": 584 }, { "clip_ratio/high_max": 0.05925852665677667, "clip_ratio/high_mean": 0.05925852665677667, "clip_ratio/low_mean": 0.019736841320991516, "clip_ratio/low_min": 0.019736841320991516, "clip_ratio/region_mean": 0.07899536797776818, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 55.5, "completions/mean_terminated_length": 55.5, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.8991814702749252, "epoch": 0.022590361445783132, "frac_reward_zero_std": 0.0, "grad_norm": 10.492538452148438, "learning_rate": 8.23030303030303e-06, "loss": 0.0176, "num_tokens": 1264918.0, "reward": 0.8827799558639526, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8827799558639526, "reward_meter_std": 0.28730660676956177, "reward_std": 0.28730660676956177, "reward_total_composite_mean": 0.8827799558639526, "reward_total_composite_std": 0.28730660676956177, "reward_total_mean": 0.8827799558639526, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8827799558639526, "rewards/meter/std": 0.28730660676956177, "rewards/total_composite/mean": 0.8827799558639526, "rewards/total_composite/std": 0.28730660676956177, "sampling/importance_sampling_ratio/max": 1.7023773193359375, "sampling/importance_sampling_ratio/mean": 1.0217286348342896, "sampling/importance_sampling_ratio/min": 0.36261942982673645, "sampling/sampling_logp_difference/max": 1.0144014358520508, "sampling/sampling_logp_difference/mean": 0.08820638805627823, "step": 585 }, { "clip_ratio/high_max": 0.054413119331002235, "clip_ratio/high_mean": 0.054413119331002235, "clip_ratio/low_mean": 0.017296989914029837, "clip_ratio/low_min": 0.017296989914029837, "clip_ratio/region_mean": 0.07171010924503207, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.9242872335016727, "epoch": 0.022628977448254557, "frac_reward_zero_std": 0.0, "grad_norm": 5.471312999725342, "learning_rate": 8.227272727272728e-06, "loss": -0.0258, "num_tokens": 1266838.0, "reward": 0.9903906583786011, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9903906583786011, "reward_meter_std": 0.01108819991350174, "reward_std": 0.011088193394243717, "reward_total_composite_mean": 0.9903906583786011, "reward_total_composite_std": 0.01108819991350174, "reward_total_mean": 0.9903906583786011, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9903906583786011, "rewards/meter/std": 0.01108819991350174, "rewards/total_composite/mean": 0.9903906583786011, "rewards/total_composite/std": 0.01108819991350174, "sampling/importance_sampling_ratio/max": 1.8207906484603882, "sampling/importance_sampling_ratio/mean": 1.0199015140533447, "sampling/importance_sampling_ratio/min": 0.4516042470932007, "sampling/sampling_logp_difference/max": 0.7949490547180176, "sampling/sampling_logp_difference/mean": 0.08893144875764847, "step": 586 }, { "clip_ratio/high_max": 0.05832049483433366, "clip_ratio/high_mean": 0.05832049483433366, "clip_ratio/low_mean": 0.04157197009772062, "clip_ratio/low_min": 0.04157197009772062, "clip_ratio/region_mean": 0.09989246493205428, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 61.25, "completions/mean_terminated_length": 61.25, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.9658423773944378, "epoch": 0.02266759345072598, "frac_reward_zero_std": 0.0, "grad_norm": 9.306632995605469, "learning_rate": 8.224242424242425e-06, "loss": 0.0326, "num_tokens": 1268600.0, "reward": 0.7887836694717407, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7887836694717407, "reward_meter_std": 0.31874561309814453, "reward_std": 0.3187456429004669, "reward_total_composite_mean": 0.7887836694717407, "reward_total_composite_std": 0.31874561309814453, "reward_total_mean": 0.7887836694717407, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7887836694717407, "rewards/meter/std": 0.31874561309814453, "rewards/total_composite/mean": 0.7887836694717407, "rewards/total_composite/std": 0.31874561309814453, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0263839960098267, "sampling/importance_sampling_ratio/min": 0.23777316510677338, "sampling/sampling_logp_difference/max": 1.4364380836486816, "sampling/sampling_logp_difference/mean": 0.10599493235349655, "step": 587 }, { "clip_ratio/high_max": 0.02804773487150669, "clip_ratio/high_mean": 0.02804773487150669, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.02804773487150669, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 390.625, "completions/mean_terminated_length": 269.25, "completions/min_length": 229.0, "completions/min_terminated_length": 229.0, "entropy": 0.43688641488552094, "epoch": 0.022706209453197405, "frac_reward_zero_std": 0.0, "grad_norm": 1.4397287368774414, "learning_rate": 8.221212121212122e-06, "loss": -0.2794, "num_tokens": 1271261.0, "reward": 0.5204770565032959, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5249999761581421, "reward_count_adherence_std": 0.2815771996974945, "reward_meter_mean": 0.9679808616638184, "reward_meter_std": 0.08165019750595093, "reward_std": 0.28585320711135864, "reward_total_composite_mean": 0.5204770565032959, "reward_total_composite_std": 0.28585317730903625, "reward_total_mean": 0.5204770565032959, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5249999761581421, "rewards/count_adherence/std": 0.2815771996974945, "rewards/meter/mean": 0.9679808616638184, "rewards/meter/std": 0.08165019750595093, "rewards/total_composite/mean": 0.5204770565032959, "rewards/total_composite/std": 0.28585317730903625, "sampling/importance_sampling_ratio/max": 1.7755378484725952, "sampling/importance_sampling_ratio/mean": 1.0239464044570923, "sampling/importance_sampling_ratio/min": 0.22953523695468903, "sampling/sampling_logp_difference/max": 1.4716987609863281, "sampling/sampling_logp_difference/mean": 0.08885669708251953, "step": 588 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.03957205289043486, "clip_ratio/low_min": 0.03957205289043486, "clip_ratio/region_mean": 0.04143772448878735, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 105.75, "completions/mean_terminated_length": 105.75, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.2631940208375454, "epoch": 0.02274482545566883, "frac_reward_zero_std": 0.0, "grad_norm": 4.359503269195557, "learning_rate": 8.21818181818182e-06, "loss": -0.0798, "num_tokens": 1273507.0, "reward": 0.7688436508178711, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9854410886764526, "reward_meter_std": 0.019673505797982216, "reward_std": 0.07495999336242676, "reward_total_composite_mean": 0.7688436508178711, "reward_total_composite_std": 0.07496000826358795, "reward_total_mean": 0.7688436508178711, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9854410886764526, "rewards/meter/std": 0.019673505797982216, "rewards/total_composite/mean": 0.7688436508178711, "rewards/total_composite/std": 0.07496000826358795, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0047855377197266, "sampling/importance_sampling_ratio/min": 0.2658828794956207, "sampling/sampling_logp_difference/max": 1.3246994018554688, "sampling/sampling_logp_difference/mean": 0.039897385984659195, "step": 589 }, { "clip_ratio/high_max": 0.029261499643325806, "clip_ratio/high_mean": 0.029261499643325806, "clip_ratio/low_mean": 0.0037128713447600603, "clip_ratio/low_min": 0.0037128713447600603, "clip_ratio/region_mean": 0.032974370988085866, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 170.375, "completions/mean_terminated_length": 121.5714340209961, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.2874904926866293, "epoch": 0.022783441458140253, "frac_reward_zero_std": 0.0, "grad_norm": 1.4748144149780273, "learning_rate": 8.215151515151517e-06, "loss": -0.1856, "num_tokens": 1275902.0, "reward": 0.6351298093795776, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.2314550280570984, "reward_meter_mean": 0.795690655708313, "reward_meter_std": 0.37391647696495056, "reward_std": 0.378958523273468, "reward_total_composite_mean": 0.6351298093795776, "reward_total_composite_std": 0.378958523273468, "reward_total_mean": 0.6351298093795776, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.2314550280570984, "rewards/meter/mean": 0.795690655708313, "rewards/meter/std": 0.37391647696495056, "rewards/total_composite/mean": 0.6351298093795776, "rewards/total_composite/std": 0.378958523273468, "sampling/importance_sampling_ratio/max": 1.574059009552002, "sampling/importance_sampling_ratio/mean": 1.005539059638977, "sampling/importance_sampling_ratio/min": 0.35161903500556946, "sampling/sampling_logp_difference/max": 1.0452070236206055, "sampling/sampling_logp_difference/mean": 0.03722504526376724, "step": 590 }, { "clip_ratio/high_max": 0.05741901881992817, "clip_ratio/high_mean": 0.05741901881992817, "clip_ratio/low_mean": 0.08319734176620841, "clip_ratio/low_min": 0.08319734176620841, "clip_ratio/region_mean": 0.14061636058613658, "completions/clipped_ratio": 0.0, "completions/max_length": 45.0, "completions/max_terminated_length": 45.0, "completions/mean_length": 40.5, "completions/mean_terminated_length": 40.5, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 1.7381385117769241, "epoch": 0.022822057460611677, "frac_reward_zero_std": 0.0, "grad_norm": 14.161763191223145, "learning_rate": 8.212121212121212e-06, "loss": 0.0618, "num_tokens": 1277586.0, "reward": 0.5279456377029419, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5279456377029419, "reward_meter_std": 0.370597779750824, "reward_std": 0.370597779750824, "reward_total_composite_mean": 0.5279456377029419, "reward_total_composite_std": 0.370597779750824, "reward_total_mean": 0.5279456377029419, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5279456377029419, "rewards/meter/std": 0.370597779750824, "rewards/total_composite/mean": 0.5279456377029419, "rewards/total_composite/std": 0.370597779750824, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0325664281845093, "sampling/importance_sampling_ratio/min": 0.31123626232147217, "sampling/sampling_logp_difference/max": 1.1672029495239258, "sampling/sampling_logp_difference/mean": 0.17503851652145386, "step": 591 }, { "clip_ratio/high_max": 0.010578885208815336, "clip_ratio/high_mean": 0.010578885208815336, "clip_ratio/low_mean": 0.0036123525351285934, "clip_ratio/low_min": 0.0036123525351285934, "clip_ratio/region_mean": 0.01419123774394393, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 389.0, "completions/mean_terminated_length": 389.0, "completions/min_length": 360.0, "completions/min_terminated_length": 360.0, "entropy": 0.12590192956849933, "epoch": 0.0228606734630831, "frac_reward_zero_std": 0.0, "grad_norm": 1.401776909828186, "learning_rate": 8.20909090909091e-06, "loss": -0.0262, "num_tokens": 1282466.0, "reward": 0.5704156160354614, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5723684430122375, "reward_count_adherence_std": 0.05215953662991524, "reward_meter_mean": 0.9966329336166382, "reward_meter_std": 0.0009374520741403103, "reward_std": 0.05163882300257683, "reward_total_composite_mean": 0.5704156160354614, "reward_total_composite_std": 0.05163882300257683, "reward_total_mean": 0.5704156160354614, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5723684430122375, "rewards/count_adherence/std": 0.05215953662991524, "rewards/meter/mean": 0.9966329336166382, "rewards/meter/std": 0.0009374520741403103, "rewards/total_composite/mean": 0.5704156160354614, "rewards/total_composite/std": 0.05163882300257683, "sampling/importance_sampling_ratio/max": 1.6324708461761475, "sampling/importance_sampling_ratio/mean": 1.001259446144104, "sampling/importance_sampling_ratio/min": 0.21806199848651886, "sampling/sampling_logp_difference/max": 1.5229759216308594, "sampling/sampling_logp_difference/mean": 0.017461512237787247, "step": 592 }, { "clip_ratio/high_max": 0.06137674581259489, "clip_ratio/high_mean": 0.06137674581259489, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/region_mean": 0.06522289966233075, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 62.375, "completions/mean_terminated_length": 62.375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.5911059454083443, "epoch": 0.022899289465554525, "frac_reward_zero_std": 0.0, "grad_norm": 7.278051853179932, "learning_rate": 8.206060606060607e-06, "loss": 0.0288, "num_tokens": 1284101.0, "reward": 0.8718899488449097, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8718899488449097, "reward_meter_std": 0.3202956020832062, "reward_std": 0.3202956020832062, "reward_total_composite_mean": 0.8718899488449097, "reward_total_composite_std": 0.3202956020832062, "reward_total_mean": 0.8718899488449097, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8718899488449097, "rewards/meter/std": 0.3202956020832062, "rewards/total_composite/mean": 0.8718899488449097, "rewards/total_composite/std": 0.3202956020832062, "sampling/importance_sampling_ratio/max": 1.7530381679534912, "sampling/importance_sampling_ratio/mean": 1.0134860277175903, "sampling/importance_sampling_ratio/min": 0.23351548612117767, "sampling/sampling_logp_difference/max": 1.4545068740844727, "sampling/sampling_logp_difference/mean": 0.06579845398664474, "step": 593 }, { "clip_ratio/high_max": 0.029324828181415796, "clip_ratio/high_mean": 0.029324828181415796, "clip_ratio/low_mean": 0.004878048785030842, "clip_ratio/low_min": 0.004878048785030842, "clip_ratio/region_mean": 0.03420287696644664, "completions/clipped_ratio": 0.0, "completions/max_length": 234.0, "completions/max_terminated_length": 234.0, "completions/mean_length": 221.75, "completions/mean_terminated_length": 221.75, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.49583121575415134, "epoch": 0.02293790546802595, "frac_reward_zero_std": 0.0, "grad_norm": 4.129243850708008, "learning_rate": 8.203030303030304e-06, "loss": -0.0155, "num_tokens": 1287579.0, "reward": 0.7827649116516113, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8928571939468384, "reward_count_adherence_std": 0.06613000482320786, "reward_meter_mean": 0.9475843906402588, "reward_meter_std": 0.13766330480575562, "reward_std": 0.32273048162460327, "reward_total_composite_mean": 0.7827649116516113, "reward_total_composite_std": 0.32273051142692566, "reward_total_mean": 0.7827649116516113, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8928571939468384, "rewards/count_adherence/std": 0.06613000482320786, "rewards/meter/mean": 0.9475843906402588, "rewards/meter/std": 0.13766330480575562, "rewards/total_composite/mean": 0.7827649116516113, "rewards/total_composite/std": 0.32273051142692566, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0077857971191406, "sampling/importance_sampling_ratio/min": 0.21689140796661377, "sampling/sampling_logp_difference/max": 1.5283584594726562, "sampling/sampling_logp_difference/mean": 0.05360095202922821, "step": 594 }, { "clip_ratio/high_max": 0.014449786627665162, "clip_ratio/high_mean": 0.014449786627665162, "clip_ratio/low_mean": 0.012580781243741512, "clip_ratio/low_min": 0.012580781243741512, "clip_ratio/region_mean": 0.027030567871406674, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 69.375, "completions/mean_terminated_length": 69.375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2160019911825657, "epoch": 0.022976521470497373, "frac_reward_zero_std": 0.0, "grad_norm": 5.483264446258545, "learning_rate": 8.2e-06, "loss": -0.0029, "num_tokens": 1289294.0, "reward": 0.825728178024292, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.825728178024292, "reward_meter_std": 0.15038101375102997, "reward_std": 0.15038102865219116, "reward_total_composite_mean": 0.825728178024292, "reward_total_composite_std": 0.15038101375102997, "reward_total_mean": 0.825728178024292, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.825728178024292, "rewards/meter/std": 0.15038101375102997, "rewards/total_composite/mean": 0.825728178024292, "rewards/total_composite/std": 0.15038101375102997, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0042104721069336, "sampling/importance_sampling_ratio/min": 0.12377584725618362, "sampling/sampling_logp_difference/max": 2.089282989501953, "sampling/sampling_logp_difference/mean": 0.043961603194475174, "step": 595 }, { "clip_ratio/high_max": 0.015763025381602347, "clip_ratio/high_mean": 0.015763025381602347, "clip_ratio/low_mean": 0.005905511789023876, "clip_ratio/low_min": 0.005905511789023876, "clip_ratio/region_mean": 0.021668537170626223, "completions/clipped_ratio": 0.0, "completions/max_length": 127.0, "completions/max_terminated_length": 127.0, "completions/mean_length": 119.875, "completions/mean_terminated_length": 119.875, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.1262477389536798, "epoch": 0.023015137472968798, "frac_reward_zero_std": 0.0, "grad_norm": 4.874330520629883, "learning_rate": 8.196969696969698e-06, "loss": 0.0213, "num_tokens": 1291573.0, "reward": 0.9766520857810974, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9766520857810974, "reward_meter_std": 0.04947783052921295, "reward_std": 0.04947783797979355, "reward_total_composite_mean": 0.9766520857810974, "reward_total_composite_std": 0.04947783052921295, "reward_total_mean": 0.9766520857810974, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9766520857810974, "rewards/meter/std": 0.04947783052921295, "rewards/total_composite/mean": 0.9766520857810974, "rewards/total_composite/std": 0.04947783052921295, "sampling/importance_sampling_ratio/max": 1.7739815711975098, "sampling/importance_sampling_ratio/mean": 1.0042794942855835, "sampling/importance_sampling_ratio/min": 0.25463753938674927, "sampling/sampling_logp_difference/max": 1.3679141998291016, "sampling/sampling_logp_difference/mean": 0.02237871289253235, "step": 596 }, { "clip_ratio/high_max": 0.037429331336170435, "clip_ratio/high_mean": 0.037429331336170435, "clip_ratio/low_mean": 0.01575682358816266, "clip_ratio/low_min": 0.01575682358816266, "clip_ratio/region_mean": 0.053186154924333096, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 62.25, "completions/mean_terminated_length": 62.25, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.4781202729791403, "epoch": 0.02305375347544022, "frac_reward_zero_std": 0.0, "grad_norm": 10.373922348022461, "learning_rate": 8.193939393939394e-06, "loss": 0.0304, "num_tokens": 1293343.0, "reward": 0.7612351775169373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7612351775169373, "reward_meter_std": 0.35920435190200806, "reward_std": 0.35920435190200806, "reward_total_composite_mean": 0.7612351775169373, "reward_total_composite_std": 0.35920435190200806, "reward_total_mean": 0.7612351775169373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7612351775169373, "rewards/meter/std": 0.35920435190200806, "rewards/total_composite/mean": 0.7612351775169373, "rewards/total_composite/std": 0.35920435190200806, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0154905319213867, "sampling/importance_sampling_ratio/min": 0.3465805947780609, "sampling/sampling_logp_difference/max": 1.0596399307250977, "sampling/sampling_logp_difference/mean": 0.0720410943031311, "step": 597 }, { "clip_ratio/high_max": 0.02251984179019928, "clip_ratio/high_mean": 0.02251984179019928, "clip_ratio/low_mean": 0.027435938362032175, "clip_ratio/low_min": 0.027435938362032175, "clip_ratio/region_mean": 0.049955780152231455, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.44570457749068737, "epoch": 0.023092369477911646, "frac_reward_zero_std": 0.0, "grad_norm": 11.545695304870605, "learning_rate": 8.190909090909091e-06, "loss": 0.0542, "num_tokens": 1294775.0, "reward": 0.9914101958274841, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9914101958274841, "reward_meter_std": 0.00670025497674942, "reward_std": 0.006700248457491398, "reward_total_composite_mean": 0.9914101958274841, "reward_total_composite_std": 0.00670025497674942, "reward_total_mean": 0.9914101958274841, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9914101958274841, "rewards/meter/std": 0.00670025497674942, "rewards/total_composite/mean": 0.9914101958274841, "rewards/total_composite/std": 0.00670025497674942, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004575490951538, "sampling/importance_sampling_ratio/min": 0.2126145362854004, "sampling/sampling_logp_difference/max": 1.5482743978500366, "sampling/sampling_logp_difference/mean": 0.06982678174972534, "step": 598 }, { "clip_ratio/high_max": 0.04812374617904425, "clip_ratio/high_mean": 0.04812374617904425, "clip_ratio/low_mean": 0.056075175292789936, "clip_ratio/low_min": 0.056075175292789936, "clip_ratio/region_mean": 0.10419892147183418, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 59.625, "completions/mean_terminated_length": 59.625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.6933459341526031, "epoch": 0.02313098548038307, "frac_reward_zero_std": 0.0, "grad_norm": 11.035534858703613, "learning_rate": 8.187878787878788e-06, "loss": 0.0205, "num_tokens": 1296492.0, "reward": 0.7003433704376221, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7003433704376221, "reward_meter_std": 0.4035126864910126, "reward_std": 0.4035126864910126, "reward_total_composite_mean": 0.7003433704376221, "reward_total_composite_std": 0.4035126864910126, "reward_total_mean": 0.7003433704376221, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7003433704376221, "rewards/meter/std": 0.4035126864910126, "rewards/total_composite/mean": 0.7003433704376221, "rewards/total_composite/std": 0.4035126864910126, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.020950198173523, "sampling/importance_sampling_ratio/min": 0.22465747594833374, "sampling/sampling_logp_difference/max": 1.493178367614746, "sampling/sampling_logp_difference/mean": 0.09578597545623779, "step": 599 }, { "clip_ratio/high_max": 0.044667589012533426, "clip_ratio/high_mean": 0.044667589012533426, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.044667589012533426, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 128.5, "completions/mean_terminated_length": 73.71428680419922, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.5339640863239765, "epoch": 0.023169601482854494, "frac_reward_zero_std": 0.0, "grad_norm": 1.266343355178833, "learning_rate": 8.184848484848486e-06, "loss": -0.1774, "num_tokens": 1298416.0, "reward": 0.8710224628448486, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9104650616645813, "reward_meter_std": 0.24043236672878265, "reward_std": 0.3519781231880188, "reward_total_composite_mean": 0.8710224628448486, "reward_total_composite_std": 0.3519780933856964, "reward_total_mean": 0.8710224628448486, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9104650616645813, "rewards/meter/std": 0.24043236672878265, "rewards/total_composite/mean": 0.8710224628448486, "rewards/total_composite/std": 0.3519780933856964, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0156185626983643, "sampling/importance_sampling_ratio/min": 0.39828726649284363, "sampling/sampling_logp_difference/max": 0.9205818176269531, "sampling/sampling_logp_difference/mean": 0.07025135308504105, "step": 600 }, { "epoch": 0.023169601482854494, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/max_length": 382.61538461538464, "eval_completions/max_terminated_length": 339.2307692307692, "eval_completions/mean_length": 205.16346153846155, "eval_completions/mean_terminated_length": 186.50274892953726, "eval_completions/min_length": 58.92307692307692, "eval_completions/min_terminated_length": 58.92307692307692, "eval_entropy": 0.2500755110612282, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 1298416.0, "eval_reward": 0.6174316452099726, "eval_reward_arabic_clean_mean": 0.9326923076923077, "eval_reward_arabic_clean_std": 0.13402181176038888, "eval_reward_count_adherence_mean": 0.9103590066616352, "eval_reward_count_adherence_std": 0.12672624450463515, "eval_reward_meter_mean": 0.6951331358689529, "eval_reward_meter_std": 0.4035135255410121, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6174316452099726, "eval_reward_total_composite_std": 0.3925590217113495, "eval_reward_total_mean": 0.6174316452099726, "eval_rewards/arabic_clean/mean": 0.9326923076923077, "eval_rewards/arabic_clean/std": 0.13402181176038888, "eval_rewards/count_adherence/mean": 0.9103590066616352, "eval_rewards/count_adherence/std": 0.12672624450463515, "eval_rewards/meter/mean": 0.6951331358689529, "eval_rewards/meter/std": 0.4035135255410121, "eval_rewards/total_composite/mean": 0.6174316452099726, "eval_rewards/total_composite/std": 0.3925590217113495, "eval_runtime": 72.8099, "eval_samples_per_second": 1.428, "eval_sampling/importance_sampling_ratio/max": 1.4073029848245473, "eval_sampling/importance_sampling_ratio/mean": 1.0067635774612427, "eval_sampling/importance_sampling_ratio/min": 0.39233631583360523, "eval_sampling/sampling_logp_difference/max": 0.9544231708233173, "eval_sampling/sampling_logp_difference/mean": 0.023065941838117745, "eval_steps_per_second": 0.179, "step": 600 }, { "clip_ratio/high_max": 0.024640035582706332, "clip_ratio/high_mean": 0.024640035582706332, "clip_ratio/low_mean": 0.015199637040495872, "clip_ratio/low_min": 0.015199637040495872, "clip_ratio/region_mean": 0.039839672623202205, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 59.125, "completions/mean_terminated_length": 59.125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.31142993830144405, "epoch": 0.023208217485325918, "frac_reward_zero_std": 0.0, "grad_norm": 4.212752342224121, "learning_rate": 8.181818181818183e-06, "loss": -0.0232, "num_tokens": 1300265.0, "reward": 0.9931149482727051, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9931149482727051, "reward_meter_std": 0.003802346298471093, "reward_std": 0.0038023567758500576, "reward_total_composite_mean": 0.9931149482727051, "reward_total_composite_std": 0.003802346298471093, "reward_total_mean": 0.9931149482727051, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9931149482727051, "rewards/meter/std": 0.003802346298471093, "rewards/total_composite/mean": 0.9931149482727051, "rewards/total_composite/std": 0.003802346298471093, "sampling/importance_sampling_ratio/max": 1.6298530101776123, "sampling/importance_sampling_ratio/mean": 1.0037661790847778, "sampling/importance_sampling_ratio/min": 0.3451637029647827, "sampling/sampling_logp_difference/max": 1.0637364387512207, "sampling/sampling_logp_difference/mean": 0.03990501910448074, "step": 601 }, { "clip_ratio/high_max": 0.029445810709148645, "clip_ratio/high_mean": 0.029445810709148645, "clip_ratio/low_mean": 0.019871795549988747, "clip_ratio/low_min": 0.019871795549988747, "clip_ratio/region_mean": 0.04931760625913739, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 157.125, "completions/mean_terminated_length": 106.42857360839844, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.3072041980922222, "epoch": 0.023246833487797342, "frac_reward_zero_std": 0.0, "grad_norm": 4.308811664581299, "learning_rate": 8.17878787878788e-06, "loss": -0.0699, "num_tokens": 1302514.0, "reward": 0.7166256308555603, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.2651650309562683, "reward_meter_mean": 0.803666353225708, "reward_meter_std": 0.3494153916835785, "reward_std": 0.3973376154899597, "reward_total_composite_mean": 0.7166256308555603, "reward_total_composite_std": 0.3973376154899597, "reward_total_mean": 0.7166256308555603, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.2651650309562683, "rewards/meter/mean": 0.803666353225708, "rewards/meter/std": 0.3494153916835785, "rewards/total_composite/mean": 0.7166256308555603, "rewards/total_composite/std": 0.3973376154899597, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0027865171432495, "sampling/importance_sampling_ratio/min": 0.14994479715824127, "sampling/sampling_logp_difference/max": 1.8974881172180176, "sampling/sampling_logp_difference/mean": 0.05126181244850159, "step": 602 }, { "clip_ratio/high_max": 0.027199887670576572, "clip_ratio/high_mean": 0.027199887670576572, "clip_ratio/low_mean": 0.006521739065647125, "clip_ratio/low_min": 0.006521739065647125, "clip_ratio/region_mean": 0.0337216267362237, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 165.75, "completions/mean_terminated_length": 116.28572082519531, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.19555421639233828, "epoch": 0.023285449490268766, "frac_reward_zero_std": 0.0, "grad_norm": 3.3487203121185303, "learning_rate": 8.175757575757577e-06, "loss": -0.1352, "num_tokens": 1304712.0, "reward": 0.7459595203399658, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.7495772838592529, "reward_meter_std": 0.4524652659893036, "reward_std": 0.45911717414855957, "reward_total_composite_mean": 0.7459595203399658, "reward_total_composite_std": 0.45911717414855957, "reward_total_mean": 0.7459595203399658, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.7495772838592529, "rewards/meter/std": 0.4524652659893036, "rewards/total_composite/mean": 0.7459595203399658, "rewards/total_composite/std": 0.45911717414855957, "sampling/importance_sampling_ratio/max": 1.7410459518432617, "sampling/importance_sampling_ratio/mean": 0.9969672560691833, "sampling/importance_sampling_ratio/min": 0.14971929788589478, "sampling/sampling_logp_difference/max": 1.8989930152893066, "sampling/sampling_logp_difference/mean": 0.034356024116277695, "step": 603 }, { "clip_ratio/high_max": 0.05441663879901171, "clip_ratio/high_mean": 0.05441663879901171, "clip_ratio/low_mean": 0.014390989672392607, "clip_ratio/low_min": 0.014390989672392607, "clip_ratio/region_mean": 0.06880762847140431, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 56.875, "completions/mean_terminated_length": 56.875, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.8086119182407856, "epoch": 0.02332406549274019, "frac_reward_zero_std": 0.0, "grad_norm": 9.974048614501953, "learning_rate": 8.172727272727273e-06, "loss": 0.0446, "num_tokens": 1306447.0, "reward": 0.8660935163497925, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8660935163497925, "reward_meter_std": 0.2912440299987793, "reward_std": 0.2912440299987793, "reward_total_composite_mean": 0.8660935163497925, "reward_total_composite_std": 0.2912440299987793, "reward_total_mean": 0.8660935163497925, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8660935163497925, "rewards/meter/std": 0.2912440299987793, "rewards/total_composite/mean": 0.8660935163497925, "rewards/total_composite/std": 0.2912440299987793, "sampling/importance_sampling_ratio/max": 1.9299688339233398, "sampling/importance_sampling_ratio/mean": 1.023914098739624, "sampling/importance_sampling_ratio/min": 0.33213749527931213, "sampling/sampling_logp_difference/max": 1.1022062301635742, "sampling/sampling_logp_difference/mean": 0.09364975243806839, "step": 604 }, { "clip_ratio/high_max": 0.008479945652652532, "clip_ratio/high_mean": 0.008479945652652532, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.008479945652652532, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 256.0, "completions/mean_length": 282.25, "completions/mean_terminated_length": 249.4285888671875, "completions/min_length": 239.0, "completions/min_terminated_length": 239.0, "entropy": 0.12214899715036154, "epoch": 0.023362681495211614, "frac_reward_zero_std": 0.0, "grad_norm": 0.45873093605041504, "learning_rate": 8.16969696969697e-06, "loss": -0.2736, "num_tokens": 1309849.0, "reward": 0.8718953132629395, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9107142686843872, "reward_count_adherence_std": 0.25253814458847046, "reward_meter_mean": 0.9013054370880127, "reward_meter_std": 0.2691211402416229, "reward_std": 0.352304071187973, "reward_total_composite_mean": 0.8718953132629395, "reward_total_composite_std": 0.352304071187973, "reward_total_mean": 0.8718953132629395, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9107142686843872, "rewards/count_adherence/std": 0.25253814458847046, "rewards/meter/mean": 0.9013054370880127, "rewards/meter/std": 0.2691211402416229, "rewards/total_composite/mean": 0.8718953132629395, "rewards/total_composite/std": 0.352304071187973, "sampling/importance_sampling_ratio/max": 1.9584522247314453, "sampling/importance_sampling_ratio/mean": 1.0060789585113525, "sampling/importance_sampling_ratio/min": 0.36774441599845886, "sampling/sampling_logp_difference/max": 1.0003671646118164, "sampling/sampling_logp_difference/mean": 0.01991911418735981, "step": 605 }, { "clip_ratio/high_max": 0.007427771924994886, "clip_ratio/high_mean": 0.007427771924994886, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007427771924994886, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 359.125, "completions/mean_terminated_length": 337.2857360839844, "completions/min_length": 323.0, "completions/min_terminated_length": 323.0, "entropy": 0.061391755007207394, "epoch": 0.02340129749768304, "frac_reward_zero_std": 0.0, "grad_norm": 0.30050450563430786, "learning_rate": 8.166666666666668e-06, "loss": -0.2859, "num_tokens": 1313978.0, "reward": 0.6420051455497742, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.644230842590332, "reward_count_adherence_std": 0.23236627876758575, "reward_meter_mean": 0.995347797870636, "reward_meter_std": 0.005631509702652693, "reward_std": 0.23203760385513306, "reward_total_composite_mean": 0.6420051455497742, "reward_total_composite_std": 0.23203761875629425, "reward_total_mean": 0.6420051455497742, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.644230842590332, "rewards/count_adherence/std": 0.23236627876758575, "rewards/meter/mean": 0.995347797870636, "rewards/meter/std": 0.005631509702652693, "rewards/total_composite/mean": 0.6420051455497742, "rewards/total_composite/std": 0.23203761875629425, "sampling/importance_sampling_ratio/max": 1.6662840843200684, "sampling/importance_sampling_ratio/mean": 1.0007063150405884, "sampling/importance_sampling_ratio/min": 0.37334179878234863, "sampling/sampling_logp_difference/max": 0.9852609634399414, "sampling/sampling_logp_difference/mean": 0.009540513157844543, "step": 606 }, { "clip_ratio/high_max": 0.00936650182120502, "clip_ratio/high_mean": 0.00936650182120502, "clip_ratio/low_mean": 0.01970675610937178, "clip_ratio/low_min": 0.01970675610937178, "clip_ratio/region_mean": 0.0290732579305768, "completions/clipped_ratio": 0.0, "completions/max_length": 175.0, "completions/max_terminated_length": 175.0, "completions/mean_length": 154.625, "completions/mean_terminated_length": 154.625, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.1557165440171957, "epoch": 0.023439913500154463, "frac_reward_zero_std": 0.0, "grad_norm": 3.355532169342041, "learning_rate": 8.163636363636365e-06, "loss": 0.0658, "num_tokens": 1316671.0, "reward": 0.4215872883796692, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.4418049454689026, "reward_meter_std": 0.4134439527988434, "reward_std": 0.38700950145721436, "reward_total_composite_mean": 0.4215872883796692, "reward_total_composite_std": 0.38700953125953674, "reward_total_mean": 0.4215872883796692, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.4418049454689026, "rewards/meter/std": 0.4134439527988434, "rewards/total_composite/mean": 0.4215872883796692, "rewards/total_composite/std": 0.38700953125953674, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0042226314544678, "sampling/importance_sampling_ratio/min": 0.1457909643650055, "sampling/sampling_logp_difference/max": 1.925581455230713, "sampling/sampling_logp_difference/mean": 0.03203287720680237, "step": 607 }, { "clip_ratio/high_max": 0.011494944454170763, "clip_ratio/high_mean": 0.011494944454170763, "clip_ratio/low_mean": 0.013556188903748989, "clip_ratio/low_min": 0.013556188903748989, "clip_ratio/region_mean": 0.025051133357919753, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 64.75, "completions/mean_terminated_length": 64.75, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.30151631124317646, "epoch": 0.023478529502625887, "frac_reward_zero_std": 0.0, "grad_norm": 6.676723957061768, "learning_rate": 8.16060606060606e-06, "loss": -0.0142, "num_tokens": 1318477.0, "reward": 0.6355655789375305, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6355655789375305, "reward_meter_std": 0.38601288199424744, "reward_std": 0.38601288199424744, "reward_total_composite_mean": 0.6355655789375305, "reward_total_composite_std": 0.38601288199424744, "reward_total_mean": 0.6355655789375305, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6355655789375305, "rewards/meter/std": 0.38601288199424744, "rewards/total_composite/mean": 0.6355655789375305, "rewards/total_composite/std": 0.38601288199424744, "sampling/importance_sampling_ratio/max": 1.405138373374939, "sampling/importance_sampling_ratio/mean": 1.007306694984436, "sampling/importance_sampling_ratio/min": 0.2254199981689453, "sampling/sampling_logp_difference/max": 1.4897899627685547, "sampling/sampling_logp_difference/mean": 0.03938113898038864, "step": 608 }, { "clip_ratio/high_max": 0.08979849983006716, "clip_ratio/high_mean": 0.08979849983006716, "clip_ratio/low_mean": 0.03448660718277097, "clip_ratio/low_min": 0.03448660718277097, "clip_ratio/region_mean": 0.12428510701283813, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 32.25, "completions/mean_terminated_length": 32.25, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.6633263751864433, "epoch": 0.02351714550509731, "frac_reward_zero_std": 0.0, "grad_norm": 11.4707612991333, "learning_rate": 8.15757575757576e-06, "loss": 0.0344, "num_tokens": 1320007.0, "reward": 0.9407675266265869, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9407675266265869, "reward_meter_std": 0.10082443803548813, "reward_std": 0.10082443803548813, "reward_total_composite_mean": 0.9407675266265869, "reward_total_composite_std": 0.10082443803548813, "reward_total_mean": 0.9407675266265869, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9407675266265869, "rewards/meter/std": 0.10082443803548813, "rewards/total_composite/mean": 0.9407675266265869, "rewards/total_composite/std": 0.10082443803548813, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0182549953460693, "sampling/importance_sampling_ratio/min": 0.16654972732067108, "sampling/sampling_logp_difference/max": 1.7924613952636719, "sampling/sampling_logp_difference/mean": 0.09047635644674301, "step": 609 }, { "clip_ratio/high_max": 0.023026442737318575, "clip_ratio/high_mean": 0.023026442737318575, "clip_ratio/low_mean": 0.019085082225501537, "clip_ratio/low_min": 0.019085082225501537, "clip_ratio/region_mean": 0.04211152496282011, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 123.75, "completions/mean_terminated_length": 68.28572082519531, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.6228674612939358, "epoch": 0.023555761507568735, "frac_reward_zero_std": 0.0, "grad_norm": 3.2543447017669678, "learning_rate": 8.154545454545455e-06, "loss": -0.0688, "num_tokens": 1321653.0, "reward": 0.547182559967041, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.5480015277862549, "reward_meter_std": 0.4364262819290161, "reward_std": 0.43759211897850037, "reward_total_composite_mean": 0.547182559967041, "reward_total_composite_std": 0.43759214878082275, "reward_total_mean": 0.547182559967041, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.5480015277862549, "rewards/meter/std": 0.4364262819290161, "rewards/total_composite/mean": 0.547182559967041, "rewards/total_composite/std": 0.43759214878082275, "sampling/importance_sampling_ratio/max": 1.6886634826660156, "sampling/importance_sampling_ratio/mean": 1.0165828466415405, "sampling/importance_sampling_ratio/min": 0.32916250824928284, "sampling/sampling_logp_difference/max": 1.111203670501709, "sampling/sampling_logp_difference/mean": 0.08188275992870331, "step": 610 }, { "clip_ratio/high_max": 0.01991767482832074, "clip_ratio/high_mean": 0.01991767482832074, "clip_ratio/low_mean": 0.012284422758966684, "clip_ratio/low_min": 0.012284422758966684, "clip_ratio/region_mean": 0.032202097587287426, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 30.75, "completions/mean_terminated_length": 30.75, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.3008997645229101, "epoch": 0.02359437751004016, "frac_reward_zero_std": 0.0, "grad_norm": 6.743978023529053, "learning_rate": 8.151515151515152e-06, "loss": -0.011, "num_tokens": 1323107.0, "reward": 0.9942895174026489, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942895174026489, "reward_meter_std": 0.001669783261604607, "reward_std": 0.001669783960096538, "reward_total_composite_mean": 0.9942895174026489, "reward_total_composite_std": 0.001669783261604607, "reward_total_mean": 0.9942895174026489, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942895174026489, "rewards/meter/std": 0.001669783261604607, "rewards/total_composite/mean": 0.9942895174026489, "rewards/total_composite/std": 0.001669783261604607, "sampling/importance_sampling_ratio/max": 1.3152281045913696, "sampling/importance_sampling_ratio/mean": 1.0020709037780762, "sampling/importance_sampling_ratio/min": 0.16944730281829834, "sampling/sampling_logp_difference/max": 1.7752132415771484, "sampling/sampling_logp_difference/mean": 0.05331292748451233, "step": 611 }, { "clip_ratio/high_max": 0.038389022229239345, "clip_ratio/high_mean": 0.038389022229239345, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/region_mean": 0.04672235599718988, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 55.625, "completions/mean_terminated_length": 55.625, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.5617792904376984, "epoch": 0.023632993512511583, "frac_reward_zero_std": 0.0, "grad_norm": 7.670464515686035, "learning_rate": 8.14848484848485e-06, "loss": 0.0336, "num_tokens": 1324792.0, "reward": 0.8599371910095215, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8599371910095215, "reward_meter_std": 0.3198857009410858, "reward_std": 0.31988564133644104, "reward_total_composite_mean": 0.8599371910095215, "reward_total_composite_std": 0.3198857009410858, "reward_total_mean": 0.8599371910095215, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8599371910095215, "rewards/meter/std": 0.3198857009410858, "rewards/total_composite/mean": 0.8599371910095215, "rewards/total_composite/std": 0.3198857009410858, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0190188884735107, "sampling/importance_sampling_ratio/min": 0.41206616163253784, "sampling/sampling_logp_difference/max": 0.9435205459594727, "sampling/sampling_logp_difference/mean": 0.06448192894458771, "step": 612 }, { "clip_ratio/high_max": 0.05149728851392865, "clip_ratio/high_mean": 0.05149728851392865, "clip_ratio/low_mean": 0.02302631549537182, "clip_ratio/low_min": 0.02302631549537182, "clip_ratio/region_mean": 0.07452360400930047, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 114.125, "completions/mean_terminated_length": 57.28571701049805, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 1.4391320049762726, "epoch": 0.023671609514983007, "frac_reward_zero_std": 0.0, "grad_norm": 3.4569411277770996, "learning_rate": 8.145454545454547e-06, "loss": -0.087, "num_tokens": 1326353.0, "reward": 0.6004748940467834, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.6199837923049927, "reward_meter_std": 0.4558931887149811, "reward_std": 0.48121726512908936, "reward_total_composite_mean": 0.6004748940467834, "reward_total_composite_std": 0.48121726512908936, "reward_total_mean": 0.6004748940467834, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.6199837923049927, "rewards/meter/std": 0.4558931887149811, "rewards/total_composite/mean": 0.6004748940467834, "rewards/total_composite/std": 0.48121726512908936, "sampling/importance_sampling_ratio/max": 1.9146498441696167, "sampling/importance_sampling_ratio/mean": 1.0191153287887573, "sampling/importance_sampling_ratio/min": 0.18310889601707458, "sampling/sampling_logp_difference/max": 1.69767427444458, "sampling/sampling_logp_difference/mean": 0.11160858720541, "step": 613 }, { "clip_ratio/high_max": 0.02200371865183115, "clip_ratio/high_mean": 0.02200371865183115, "clip_ratio/low_mean": 0.022413392609450966, "clip_ratio/low_min": 0.022413392609450966, "clip_ratio/region_mean": 0.044417111261282116, "completions/clipped_ratio": 0.0, "completions/max_length": 158.0, "completions/max_terminated_length": 158.0, "completions/mean_length": 122.625, "completions/mean_terminated_length": 122.625, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.29260148853063583, "epoch": 0.02371022551745443, "frac_reward_zero_std": 0.0, "grad_norm": 5.5195722579956055, "learning_rate": 8.142424242424242e-06, "loss": -0.025, "num_tokens": 1328678.0, "reward": 0.48351413011550903, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_meter_mean": 0.5231146216392517, "reward_meter_std": 0.36290431022644043, "reward_std": 0.3324859142303467, "reward_total_composite_mean": 0.48351413011550903, "reward_total_composite_std": 0.3324858844280243, "reward_total_mean": 0.48351413011550903, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/meter/mean": 0.5231146216392517, "rewards/meter/std": 0.36290431022644043, "rewards/total_composite/mean": 0.48351413011550903, "rewards/total_composite/std": 0.3324858844280243, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0053133964538574, "sampling/importance_sampling_ratio/min": 0.047882918268442154, "sampling/sampling_logp_difference/max": 3.038996458053589, "sampling/sampling_logp_difference/mean": 0.05160455033183098, "step": 614 }, { "clip_ratio/high_max": 0.018577482085675, "clip_ratio/high_mean": 0.018577482085675, "clip_ratio/low_mean": 0.014563622185960412, "clip_ratio/low_min": 0.014563622185960412, "clip_ratio/region_mean": 0.03314110427163541, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 105.25, "completions/mean_terminated_length": 105.25, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.42340752109885216, "epoch": 0.023748841519925856, "frac_reward_zero_std": 0.0, "grad_norm": 4.773278713226318, "learning_rate": 8.139393939393941e-06, "loss": 0.016, "num_tokens": 1330808.0, "reward": 0.7684873342514038, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7684873342514038, "reward_meter_std": 0.3484794795513153, "reward_std": 0.3484795093536377, "reward_total_composite_mean": 0.7684873342514038, "reward_total_composite_std": 0.3484794795513153, "reward_total_mean": 0.7684873342514038, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7684873342514038, "rewards/meter/std": 0.3484794795513153, "rewards/total_composite/mean": 0.7684873342514038, "rewards/total_composite/std": 0.3484794795513153, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.013346791267395, "sampling/importance_sampling_ratio/min": 0.3099987804889679, "sampling/sampling_logp_difference/max": 1.171186923980713, "sampling/sampling_logp_difference/mean": 0.054694000631570816, "step": 615 }, { "clip_ratio/high_max": 0.0064004448358900845, "clip_ratio/high_mean": 0.0064004448358900845, "clip_ratio/low_mean": 0.004212898784317076, "clip_ratio/low_min": 0.004212898784317076, "clip_ratio/region_mean": 0.01061334362020716, "completions/clipped_ratio": 0.0, "completions/max_length": 210.0, "completions/max_terminated_length": 210.0, "completions/mean_length": 198.75, "completions/mean_terminated_length": 198.75, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.05790994456037879, "epoch": 0.023787457522397283, "frac_reward_zero_std": 0.0, "grad_norm": 2.1530308723449707, "learning_rate": 8.136363636363637e-06, "loss": 0.0206, "num_tokens": 1333982.0, "reward": 0.9137976765632629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.9345196485519409, "reward_meter_std": 0.17161889374256134, "reward_std": 0.1733204573392868, "reward_total_composite_mean": 0.9137976765632629, "reward_total_composite_std": 0.1733204871416092, "reward_total_mean": 0.9137976765632629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.9345196485519409, "rewards/meter/std": 0.17161889374256134, "rewards/total_composite/mean": 0.9137976765632629, "rewards/total_composite/std": 0.1733204871416092, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017095804214478, "sampling/importance_sampling_ratio/min": 0.30434298515319824, "sampling/sampling_logp_difference/max": 1.1895999908447266, "sampling/sampling_logp_difference/mean": 0.009508727118372917, "step": 616 }, { "clip_ratio/high_max": 0.013311688497196883, "clip_ratio/high_mean": 0.013311688497196883, "clip_ratio/low_mean": 0.026029597967863083, "clip_ratio/low_min": 0.026029597967863083, "clip_ratio/region_mean": 0.039341286465059966, "completions/clipped_ratio": 0.0, "completions/max_length": 456.0, "completions/max_terminated_length": 456.0, "completions/mean_length": 189.375, "completions/mean_terminated_length": 189.375, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.9195150500163436, "epoch": 0.023826073524868707, "frac_reward_zero_std": 0.0, "grad_norm": 3.8700666427612305, "learning_rate": 8.133333333333334e-06, "loss": 0.1404, "num_tokens": 1337009.0, "reward": 0.11995942890644073, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.17251639068126678, "reward_meter_mean": 0.12910132110118866, "reward_meter_std": 0.23855595290660858, "reward_std": 0.22783413529396057, "reward_total_composite_mean": 0.11995942890644073, "reward_total_composite_std": 0.22783412039279938, "reward_total_mean": 0.11995942890644073, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.17251639068126678, "rewards/meter/mean": 0.12910132110118866, "rewards/meter/std": 0.23855595290660858, "rewards/total_composite/mean": 0.11995942890644073, "rewards/total_composite/std": 0.22783412039279938, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0223716497421265, "sampling/importance_sampling_ratio/min": 0.009397400543093681, "sampling/sampling_logp_difference/max": 4.667322158813477, "sampling/sampling_logp_difference/mean": 0.08845033496618271, "step": 617 }, { "clip_ratio/high_max": 0.031583026982843876, "clip_ratio/high_mean": 0.031583026982843876, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.031583026982843876, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 339.375, "completions/mean_terminated_length": 51.66666793823242, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.19933610782027245, "epoch": 0.02386468952734013, "frac_reward_zero_std": 0.0, "grad_norm": 0.84645676612854, "learning_rate": 8.130303030303031e-06, "loss": -0.0697, "num_tokens": 1338388.0, "reward": 0.37216296792030334, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_meter_mean": 0.5469461679458618, "reward_meter_std": 0.45063626766204834, "reward_std": 0.5076145529747009, "reward_total_composite_mean": 0.37216296792030334, "reward_total_composite_std": 0.5076146125793457, "reward_total_mean": 0.37216296792030334, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/meter/mean": 0.5469461679458618, "rewards/meter/std": 0.45063626766204834, "rewards/total_composite/mean": 0.37216296792030334, "rewards/total_composite/std": 0.5076146125793457, "sampling/importance_sampling_ratio/max": 1.5025615692138672, "sampling/importance_sampling_ratio/mean": 1.009710669517517, "sampling/importance_sampling_ratio/min": 0.24296501278877258, "sampling/sampling_logp_difference/max": 1.4148378372192383, "sampling/sampling_logp_difference/mean": 0.07076752930879593, "step": 618 }, { "clip_ratio/high_max": 0.0026461693923920393, "clip_ratio/high_mean": 0.0026461693923920393, "clip_ratio/low_mean": 0.014778794953599572, "clip_ratio/low_min": 0.014778794953599572, "clip_ratio/region_mean": 0.01742496434599161, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 92.625, "completions/mean_terminated_length": 92.625, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.1813750467263162, "epoch": 0.023903305529811555, "frac_reward_zero_std": 0.0, "grad_norm": 4.1318182945251465, "learning_rate": 8.127272727272728e-06, "loss": -0.0004, "num_tokens": 1340369.0, "reward": 0.7933108806610107, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7933108806610107, "reward_meter_std": 0.2967395782470703, "reward_std": 0.2967395484447479, "reward_total_composite_mean": 0.7933108806610107, "reward_total_composite_std": 0.2967395782470703, "reward_total_mean": 0.7933108806610107, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7933108806610107, "rewards/meter/std": 0.2967395782470703, "rewards/total_composite/mean": 0.7933108806610107, "rewards/total_composite/std": 0.2967395782470703, "sampling/importance_sampling_ratio/max": 1.3991177082061768, "sampling/importance_sampling_ratio/mean": 1.001517415046692, "sampling/importance_sampling_ratio/min": 0.37277644872665405, "sampling/sampling_logp_difference/max": 0.9867763519287109, "sampling/sampling_logp_difference/mean": 0.026417352259159088, "step": 619 }, { "clip_ratio/high_max": 0.0851530022919178, "clip_ratio/high_mean": 0.0851530022919178, "clip_ratio/low_mean": 0.011140820104628801, "clip_ratio/low_min": 0.011140820104628801, "clip_ratio/region_mean": 0.0962938223965466, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 33.875, "completions/mean_terminated_length": 33.875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.5899157114326954, "epoch": 0.02394192153228298, "frac_reward_zero_std": 0.0, "grad_norm": 11.696823120117188, "learning_rate": 8.124242424242424e-06, "loss": 0.0358, "num_tokens": 1341808.0, "reward": 0.978384256362915, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.978384256362915, "reward_meter_std": 0.029845381155610085, "reward_std": 0.029845381155610085, "reward_total_composite_mean": 0.978384256362915, "reward_total_composite_std": 0.029845381155610085, "reward_total_mean": 0.978384256362915, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.978384256362915, "rewards/meter/std": 0.029845381155610085, "rewards/total_composite/mean": 0.978384256362915, "rewards/total_composite/std": 0.029845381155610085, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007791519165039, "sampling/importance_sampling_ratio/min": 0.1091901957988739, "sampling/sampling_logp_difference/max": 2.2146639823913574, "sampling/sampling_logp_difference/mean": 0.11957071721553802, "step": 620 }, { "clip_ratio/high_max": 0.017727577593177557, "clip_ratio/high_mean": 0.017727577593177557, "clip_ratio/low_mean": 0.011628984240815043, "clip_ratio/low_min": 0.011628984240815043, "clip_ratio/region_mean": 0.0293565618339926, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 72.375, "completions/mean_terminated_length": 72.375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2109714262187481, "epoch": 0.023980537534754404, "frac_reward_zero_std": 0.0, "grad_norm": 4.917610168457031, "learning_rate": 8.121212121212121e-06, "loss": 0.034, "num_tokens": 1343571.0, "reward": 0.9518520832061768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9518520832061768, "reward_meter_std": 0.032847389578819275, "reward_std": 0.032847389578819275, "reward_total_composite_mean": 0.9518520832061768, "reward_total_composite_std": 0.032847389578819275, "reward_total_mean": 0.9518520832061768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9518520832061768, "rewards/meter/std": 0.032847389578819275, "rewards/total_composite/mean": 0.9518520832061768, "rewards/total_composite/std": 0.032847389578819275, "sampling/importance_sampling_ratio/max": 1.9684488773345947, "sampling/importance_sampling_ratio/mean": 1.0048478841781616, "sampling/importance_sampling_ratio/min": 0.06061416491866112, "sampling/sampling_logp_difference/max": 2.8032267093658447, "sampling/sampling_logp_difference/mean": 0.04176875576376915, "step": 621 }, { "clip_ratio/high_max": 0.03944407729431987, "clip_ratio/high_mean": 0.03944407729431987, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/region_mean": 0.045694077387452126, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 113.125, "completions/mean_terminated_length": 56.142860412597656, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.4394574910402298, "epoch": 0.024019153537225828, "frac_reward_zero_std": 0.0, "grad_norm": 1.2342267036437988, "learning_rate": 8.118181818181819e-06, "loss": -0.1456, "num_tokens": 1345212.0, "reward": 0.8246999979019165, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8251349329948425, "reward_meter_std": 0.3412078320980072, "reward_std": 0.342404842376709, "reward_total_composite_mean": 0.8246999979019165, "reward_total_composite_std": 0.342404842376709, "reward_total_mean": 0.8246999979019165, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8251349329948425, "rewards/meter/std": 0.3412078320980072, "rewards/total_composite/mean": 0.8246999979019165, "rewards/total_composite/std": 0.342404842376709, "sampling/importance_sampling_ratio/max": 1.750449538230896, "sampling/importance_sampling_ratio/mean": 1.0137029886245728, "sampling/importance_sampling_ratio/min": 0.22795508801937103, "sampling/sampling_logp_difference/max": 1.4786067008972168, "sampling/sampling_logp_difference/mean": 0.07290157675743103, "step": 622 }, { "clip_ratio/high_max": 0.025720802485011518, "clip_ratio/high_mean": 0.025720802485011518, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/region_mean": 0.027344179106876254, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 125.125, "completions/mean_terminated_length": 69.85714721679688, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.31833127327263355, "epoch": 0.024057769539697252, "frac_reward_zero_std": 0.0, "grad_norm": 1.7279670238494873, "learning_rate": 8.115151515151516e-06, "loss": -0.1006, "num_tokens": 1346949.0, "reward": 0.7471871376037598, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7928895950317383, "reward_meter_std": 0.3771921992301941, "reward_std": 0.4512399137020111, "reward_total_composite_mean": 0.7471871376037598, "reward_total_composite_std": 0.4512399435043335, "reward_total_mean": 0.7471871376037598, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7928895950317383, "rewards/meter/std": 0.3771921992301941, "rewards/total_composite/mean": 0.7471871376037598, "rewards/total_composite/std": 0.4512399435043335, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0057697296142578, "sampling/importance_sampling_ratio/min": 0.3635649085044861, "sampling/sampling_logp_difference/max": 1.0117974281311035, "sampling/sampling_logp_difference/mean": 0.050993990153074265, "step": 623 }, { "clip_ratio/high_max": 0.0006510416860692203, "clip_ratio/high_mean": 0.0006510416860692203, "clip_ratio/low_mean": 0.0013736264081671834, "clip_ratio/low_min": 0.0013736264081671834, "clip_ratio/region_mean": 0.0020246680942364037, "completions/clipped_ratio": 0.0, "completions/max_length": 192.0, "completions/max_terminated_length": 192.0, "completions/mean_length": 182.75, "completions/mean_terminated_length": 182.75, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.04385493486188352, "epoch": 0.024096385542168676, "frac_reward_zero_std": 0.0, "grad_norm": 2.252504825592041, "learning_rate": 8.112121212121213e-06, "loss": -0.0178, "num_tokens": 1350067.0, "reward": 0.9957711100578308, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957711100578308, "reward_meter_std": 0.0014193379320204258, "reward_std": 0.0014193379320204258, "reward_total_composite_mean": 0.9957711100578308, "reward_total_composite_std": 0.0014193379320204258, "reward_total_mean": 0.9957711100578308, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957711100578308, "rewards/meter/std": 0.0014193379320204258, "rewards/total_composite/mean": 0.9957711100578308, "rewards/total_composite/std": 0.0014193379320204258, "sampling/importance_sampling_ratio/max": 1.406175971031189, "sampling/importance_sampling_ratio/mean": 1.0003677606582642, "sampling/importance_sampling_ratio/min": 0.5395426154136658, "sampling/sampling_logp_difference/max": 0.6170334815979004, "sampling/sampling_logp_difference/mean": 0.005115547217428684, "step": 624 }, { "clip_ratio/high_max": 0.00999945483636111, "clip_ratio/high_mean": 0.00999945483636111, "clip_ratio/low_mean": 0.005678939749486744, "clip_ratio/low_min": 0.005678939749486744, "clip_ratio/region_mean": 0.015678394585847855, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 297.0, "completions/mean_terminated_length": 266.2857360839844, "completions/min_length": 247.0, "completions/min_terminated_length": 247.0, "entropy": 0.09663973702117801, "epoch": 0.0241350015446401, "frac_reward_zero_std": 0.0, "grad_norm": 2.2000861167907715, "learning_rate": 8.10909090909091e-06, "loss": -0.1074, "num_tokens": 1353587.0, "reward": 0.4286315441131592, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9027777910232544, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.5770298838615417, "reward_meter_std": 0.47208163142204285, "reward_std": 0.44896507263183594, "reward_total_composite_mean": 0.4286315441131592, "reward_total_composite_std": 0.44896507263183594, "reward_total_mean": 0.4286315441131592, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9027777910232544, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.5770298838615417, "rewards/meter/std": 0.47208163142204285, "rewards/total_composite/mean": 0.4286315441131592, "rewards/total_composite/std": 0.44896507263183594, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0011159181594849, "sampling/importance_sampling_ratio/min": 0.22995755076408386, "sampling/sampling_logp_difference/max": 1.469860553741455, "sampling/sampling_logp_difference/mean": 0.021907327696681023, "step": 625 }, { "clip_ratio/high_max": 0.002810997946653515, "clip_ratio/high_mean": 0.002810997946653515, "clip_ratio/low_mean": 0.005793766817077994, "clip_ratio/low_min": 0.005793766817077994, "clip_ratio/region_mean": 0.00860476476373151, "completions/clipped_ratio": 0.0, "completions/max_length": 272.0, "completions/max_terminated_length": 272.0, "completions/mean_length": 258.25, "completions/mean_terminated_length": 258.25, "completions/min_length": 239.0, "completions/min_terminated_length": 239.0, "entropy": 0.08377446280792356, "epoch": 0.024173617547111524, "frac_reward_zero_std": 0.0, "grad_norm": 1.3433260917663574, "learning_rate": 8.106060606060606e-06, "loss": -0.0186, "num_tokens": 1357445.0, "reward": 0.9136239886283875, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.921875, "reward_count_adherence_std": 0.06469365209341049, "reward_meter_mean": 0.9910991787910461, "reward_meter_std": 0.006634681485593319, "reward_std": 0.06364713609218597, "reward_total_composite_mean": 0.9136239886283875, "reward_total_composite_std": 0.06364713609218597, "reward_total_mean": 0.9136239886283875, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.921875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/meter/mean": 0.9910991787910461, "rewards/meter/std": 0.006634681485593319, "rewards/total_composite/mean": 0.9136239886283875, "rewards/total_composite/std": 0.06364713609218597, "sampling/importance_sampling_ratio/max": 1.596511960029602, "sampling/importance_sampling_ratio/mean": 1.0006234645843506, "sampling/importance_sampling_ratio/min": 0.2601662278175354, "sampling/sampling_logp_difference/max": 1.3464345932006836, "sampling/sampling_logp_difference/mean": 0.01168814580887556, "step": 626 }, { "clip_ratio/high_max": 0.04422784689813852, "clip_ratio/high_mean": 0.04422784689813852, "clip_ratio/low_mean": 0.010802469216287136, "clip_ratio/low_min": 0.010802469216287136, "clip_ratio/region_mean": 0.05503031611442566, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 175.25, "completions/mean_terminated_length": 63.0, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.9980961419641972, "epoch": 0.02421223354958295, "frac_reward_zero_std": 0.0, "grad_norm": 1.754539966583252, "learning_rate": 8.103030303030303e-06, "loss": -0.1003, "num_tokens": 1359055.0, "reward": 0.5797286033630371, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.25877460837364197, "reward_meter_mean": 0.7318032383918762, "reward_meter_std": 0.3413369357585907, "reward_std": 0.4377219080924988, "reward_total_composite_mean": 0.5797286033630371, "reward_total_composite_std": 0.4377219080924988, "reward_total_mean": 0.5797286033630371, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.25877460837364197, "rewards/meter/mean": 0.7318032383918762, "rewards/meter/std": 0.3413369357585907, "rewards/total_composite/mean": 0.5797286033630371, "rewards/total_composite/std": 0.4377219080924988, "sampling/importance_sampling_ratio/max": 1.7900029420852661, "sampling/importance_sampling_ratio/mean": 1.012277603149414, "sampling/importance_sampling_ratio/min": 0.25523537397384644, "sampling/sampling_logp_difference/max": 1.3655691146850586, "sampling/sampling_logp_difference/mean": 0.11965037882328033, "step": 627 }, { "clip_ratio/high_max": 0.020865230937488377, "clip_ratio/high_mean": 0.020865230937488377, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/region_mean": 0.027809675433672965, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 92.125, "completions/mean_terminated_length": 92.125, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.19556331355124712, "epoch": 0.024250849552054372, "frac_reward_zero_std": 0.0, "grad_norm": 5.110913276672363, "learning_rate": 8.1e-06, "loss": 0.0687, "num_tokens": 1361096.0, "reward": 0.970349907875061, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.970349907875061, "reward_meter_std": 0.06832630932331085, "reward_std": 0.06832629442214966, "reward_total_composite_mean": 0.970349907875061, "reward_total_composite_std": 0.06832630932331085, "reward_total_mean": 0.970349907875061, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.970349907875061, "rewards/meter/std": 0.06832630932331085, "rewards/total_composite/mean": 0.970349907875061, "rewards/total_composite/std": 0.06832630932331085, "sampling/importance_sampling_ratio/max": 1.841081142425537, "sampling/importance_sampling_ratio/mean": 0.9971374273300171, "sampling/importance_sampling_ratio/min": 0.31275883316993713, "sampling/sampling_logp_difference/max": 1.1623228788375854, "sampling/sampling_logp_difference/mean": 0.02751935087144375, "step": 628 }, { "clip_ratio/high_max": 0.016275564790703356, "clip_ratio/high_mean": 0.016275564790703356, "clip_ratio/low_mean": 0.010869565419852734, "clip_ratio/low_min": 0.010869565419852734, "clip_ratio/region_mean": 0.02714513021055609, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 182.875, "completions/mean_terminated_length": 73.16667175292969, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.30349646881222725, "epoch": 0.024289465554525796, "frac_reward_zero_std": 0.0, "grad_norm": 0.993527352809906, "learning_rate": 8.096969696969698e-06, "loss": -0.1464, "num_tokens": 1362727.0, "reward": 0.662638783454895, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9057246446609497, "reward_meter_std": 0.21483053267002106, "reward_std": 0.46009397506713867, "reward_total_composite_mean": 0.662638783454895, "reward_total_composite_std": 0.46009400486946106, "reward_total_mean": 0.662638783454895, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9057246446609497, "rewards/meter/std": 0.21483053267002106, "rewards/total_composite/mean": 0.662638783454895, "rewards/total_composite/std": 0.46009400486946106, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0186997652053833, "sampling/importance_sampling_ratio/min": 0.46451571583747864, "sampling/sampling_logp_difference/max": 0.8761682510375977, "sampling/sampling_logp_difference/mean": 0.0446912907063961, "step": 629 }, { "clip_ratio/high_max": 0.04845884535461664, "clip_ratio/high_mean": 0.04845884535461664, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.05236509535461664, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.0, "completions/mean_terminated_length": 62.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.25993255618959665, "epoch": 0.02432808155699722, "frac_reward_zero_std": 0.0, "grad_norm": 8.351807594299316, "learning_rate": 8.093939393939395e-06, "loss": 0.0194, "num_tokens": 1364463.0, "reward": 0.9843227863311768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9843227863311768, "reward_meter_std": 0.027253830805420876, "reward_std": 0.027253834530711174, "reward_total_composite_mean": 0.9843227863311768, "reward_total_composite_std": 0.027253830805420876, "reward_total_mean": 0.9843227863311768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9843227863311768, "rewards/meter/std": 0.027253830805420876, "rewards/total_composite/mean": 0.9843227863311768, "rewards/total_composite/std": 0.027253830805420876, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0087597370147705, "sampling/importance_sampling_ratio/min": 0.24401849508285522, "sampling/sampling_logp_difference/max": 1.4105112552642822, "sampling/sampling_logp_difference/mean": 0.05221335589885712, "step": 630 }, { "clip_ratio/high_max": 0.011983279720880091, "clip_ratio/high_mean": 0.011983279720880091, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.011983279720880091, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 127.25, "completions/mean_terminated_length": 72.28572082519531, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.1889364793896675, "epoch": 0.024366697559468645, "frac_reward_zero_std": 0.0, "grad_norm": 0.8814429640769958, "learning_rate": 8.090909090909092e-06, "loss": -0.1755, "num_tokens": 1366217.0, "reward": 0.8639343976974487, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.8845906257629395, "reward_meter_std": 0.2908637821674347, "reward_std": 0.3492541015148163, "reward_total_composite_mean": 0.8639343976974487, "reward_total_composite_std": 0.3492541015148163, "reward_total_mean": 0.8639343976974487, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.8845906257629395, "rewards/meter/std": 0.2908637821674347, "rewards/total_composite/mean": 0.8639343976974487, "rewards/total_composite/std": 0.3492541015148163, "sampling/importance_sampling_ratio/max": 1.5585085153579712, "sampling/importance_sampling_ratio/mean": 1.006989598274231, "sampling/importance_sampling_ratio/min": 0.4087888300418854, "sampling/sampling_logp_difference/max": 0.8945565223693848, "sampling/sampling_logp_difference/mean": 0.028667865321040154, "step": 631 }, { "clip_ratio/high_max": 0.007443342707119882, "clip_ratio/high_mean": 0.007443342707119882, "clip_ratio/low_mean": 0.0009259259095415473, "clip_ratio/low_min": 0.0009259259095415473, "clip_ratio/region_mean": 0.00836926861666143, "completions/clipped_ratio": 0.0, "completions/max_length": 135.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 111.0, "completions/mean_terminated_length": 111.0, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.07266335096210241, "epoch": 0.02440531356194007, "frac_reward_zero_std": 0.0, "grad_norm": 3.513773202896118, "learning_rate": 8.08787878787879e-06, "loss": 0.0991, "num_tokens": 1368537.0, "reward": 0.9886435270309448, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9886435270309448, "reward_meter_std": 0.0022036279551684856, "reward_std": 0.0022036198060959578, "reward_total_composite_mean": 0.9886435270309448, "reward_total_composite_std": 0.0022036279551684856, "reward_total_mean": 0.9886435270309448, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9886435270309448, "rewards/meter/std": 0.0022036279551684856, "rewards/total_composite/mean": 0.9886435270309448, "rewards/total_composite/std": 0.0022036279551684856, "sampling/importance_sampling_ratio/max": 1.52896249294281, "sampling/importance_sampling_ratio/mean": 1.0014369487762451, "sampling/importance_sampling_ratio/min": 0.39802050590515137, "sampling/sampling_logp_difference/max": 0.9212517142295837, "sampling/sampling_logp_difference/mean": 0.01080810371786356, "step": 632 }, { "clip_ratio/high_max": 0.022631968837231398, "clip_ratio/high_mean": 0.022631968837231398, "clip_ratio/low_mean": 0.0013020833721384406, "clip_ratio/low_min": 0.0013020833721384406, "clip_ratio/region_mean": 0.023934052209369838, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 84.875, "completions/mean_terminated_length": 84.875, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.17416242323815823, "epoch": 0.024443929564411493, "frac_reward_zero_std": 0.0, "grad_norm": 3.3415863513946533, "learning_rate": 8.084848484848485e-06, "loss": 0.0485, "num_tokens": 1370640.0, "reward": 0.8842945098876953, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8842945098876953, "reward_meter_std": 0.31306833028793335, "reward_std": 0.31306830048561096, "reward_total_composite_mean": 0.8842945098876953, "reward_total_composite_std": 0.31306833028793335, "reward_total_mean": 0.8842945098876953, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8842945098876953, "rewards/meter/std": 0.31306833028793335, "rewards/total_composite/mean": 0.8842945098876953, "rewards/total_composite/std": 0.31306833028793335, "sampling/importance_sampling_ratio/max": 1.9100323915481567, "sampling/importance_sampling_ratio/mean": 1.005111813545227, "sampling/importance_sampling_ratio/min": 0.2329607456922531, "sampling/sampling_logp_difference/max": 1.4568853378295898, "sampling/sampling_logp_difference/mean": 0.02020621858537197, "step": 633 }, { "clip_ratio/high_max": 0.018129791948013008, "clip_ratio/high_mean": 0.018129791948013008, "clip_ratio/low_mean": 0.012231739703565836, "clip_ratio/low_min": 0.012231739703565836, "clip_ratio/region_mean": 0.030361531651578844, "completions/clipped_ratio": 0.0, "completions/max_length": 142.0, "completions/max_terminated_length": 142.0, "completions/mean_length": 126.25, "completions/mean_terminated_length": 126.25, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.20136078912764788, "epoch": 0.024482545566882917, "frac_reward_zero_std": 0.0, "grad_norm": 2.963343858718872, "learning_rate": 8.081818181818182e-06, "loss": 0.0184, "num_tokens": 1372994.0, "reward": 0.9853245615959167, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9853245615959167, "reward_meter_std": 0.021229863166809082, "reward_std": 0.02122986875474453, "reward_total_composite_mean": 0.9853245615959167, "reward_total_composite_std": 0.021229863166809082, "reward_total_mean": 0.9853245615959167, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9853245615959167, "rewards/meter/std": 0.021229863166809082, "rewards/total_composite/mean": 0.9853245615959167, "rewards/total_composite/std": 0.021229863166809082, "sampling/importance_sampling_ratio/max": 1.6705741882324219, "sampling/importance_sampling_ratio/mean": 1.0044063329696655, "sampling/importance_sampling_ratio/min": 0.3048539459705353, "sampling/sampling_logp_difference/max": 1.187922477722168, "sampling/sampling_logp_difference/mean": 0.027480771765112877, "step": 634 }, { "clip_ratio/high_max": 0.012339744134806097, "clip_ratio/high_mean": 0.012339744134806097, "clip_ratio/low_mean": 0.007863856852054596, "clip_ratio/low_min": 0.007863856852054596, "clip_ratio/region_mean": 0.020203600986860693, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 61.875, "completions/mean_terminated_length": 61.875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.11720475321635604, "epoch": 0.02452116156935434, "frac_reward_zero_std": 0.0, "grad_norm": 6.4878387451171875, "learning_rate": 8.07878787878788e-06, "loss": -0.0067, "num_tokens": 1374785.0, "reward": 0.9954343438148499, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9954343438148499, "reward_meter_std": 0.0015961348544806242, "reward_std": 0.0015961244935169816, "reward_total_composite_mean": 0.9954343438148499, "reward_total_composite_std": 0.0015961348544806242, "reward_total_mean": 0.9954343438148499, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9954343438148499, "rewards/meter/std": 0.0015961348544806242, "rewards/total_composite/mean": 0.9954343438148499, "rewards/total_composite/std": 0.0015961348544806242, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9988678693771362, "sampling/importance_sampling_ratio/min": 0.15966175496578217, "sampling/sampling_logp_difference/max": 1.8346977233886719, "sampling/sampling_logp_difference/mean": 0.0258852057158947, "step": 635 }, { "clip_ratio/high_max": 0.002884615445509553, "clip_ratio/high_mean": 0.002884615445509553, "clip_ratio/low_mean": 0.0020576479146257043, "clip_ratio/low_min": 0.0020576479146257043, "clip_ratio/region_mean": 0.004942263360135257, "completions/clipped_ratio": 0.0, "completions/max_length": 130.0, "completions/max_terminated_length": 130.0, "completions/mean_length": 122.25, "completions/mean_terminated_length": 122.25, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.03814642573706806, "epoch": 0.024559777571825765, "frac_reward_zero_std": 0.0, "grad_norm": 1.9705545902252197, "learning_rate": 8.075757575757577e-06, "loss": -0.0179, "num_tokens": 1377307.0, "reward": 0.9956600069999695, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956600069999695, "reward_meter_std": 0.001489842776209116, "reward_std": 0.001489846152253449, "reward_total_composite_mean": 0.9956600069999695, "reward_total_composite_std": 0.001489842776209116, "reward_total_mean": 0.9956600069999695, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956600069999695, "rewards/meter/std": 0.001489842776209116, "rewards/total_composite/mean": 0.9956600069999695, "rewards/total_composite/std": 0.001489842776209116, "sampling/importance_sampling_ratio/max": 1.2143502235412598, "sampling/importance_sampling_ratio/mean": 1.0008108615875244, "sampling/importance_sampling_ratio/min": 0.10169878602027893, "sampling/sampling_logp_difference/max": 2.2857398986816406, "sampling/sampling_logp_difference/mean": 0.0057179187424480915, "step": 636 }, { "clip_ratio/high_max": 0.010326079092919827, "clip_ratio/high_mean": 0.010326079092919827, "clip_ratio/low_mean": 0.003179112682119012, "clip_ratio/low_min": 0.003179112682119012, "clip_ratio/region_mean": 0.013505191775038838, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 123.5, "completions/mean_terminated_length": 123.5, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.18402807414531708, "epoch": 0.02459839357429719, "frac_reward_zero_std": 0.0, "grad_norm": 2.965711832046509, "learning_rate": 8.072727272727274e-06, "loss": 0.0233, "num_tokens": 1379671.0, "reward": 0.9958841800689697, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958841800689697, "reward_meter_std": 0.0024583181366324425, "reward_std": 0.002458317205309868, "reward_total_composite_mean": 0.9958841800689697, "reward_total_composite_std": 0.0024583181366324425, "reward_total_mean": 0.9958841800689697, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958841800689697, "rewards/meter/std": 0.0024583181366324425, "rewards/total_composite/mean": 0.9958841800689697, "rewards/total_composite/std": 0.0024583181366324425, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0019030570983887, "sampling/importance_sampling_ratio/min": 0.38929271697998047, "sampling/sampling_logp_difference/max": 0.9434237480163574, "sampling/sampling_logp_difference/mean": 0.027591824531555176, "step": 637 }, { "clip_ratio/high_max": 0.051263275323435664, "clip_ratio/high_mean": 0.051263275323435664, "clip_ratio/low_mean": 0.023809524718672037, "clip_ratio/low_min": 0.023809524718672037, "clip_ratio/region_mean": 0.0750728000421077, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 64.375, "completions/mean_terminated_length": 64.375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.5689874831587076, "epoch": 0.024637009576768613, "frac_reward_zero_std": 0.0, "grad_norm": 11.34463882446289, "learning_rate": 8.069696969696971e-06, "loss": -0.0238, "num_tokens": 1381546.0, "reward": 0.9862507581710815, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9862507581710815, "reward_meter_std": 0.01588655635714531, "reward_std": 0.015886547043919563, "reward_total_composite_mean": 0.9862507581710815, "reward_total_composite_std": 0.01588655635714531, "reward_total_mean": 0.9862507581710815, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9862507581710815, "rewards/meter/std": 0.01588655635714531, "rewards/total_composite/mean": 0.9862507581710815, "rewards/total_composite/std": 0.01588655635714531, "sampling/importance_sampling_ratio/max": 1.9803534746170044, "sampling/importance_sampling_ratio/mean": 1.005232810974121, "sampling/importance_sampling_ratio/min": 0.05610604211688042, "sampling/sampling_logp_difference/max": 2.88051176071167, "sampling/sampling_logp_difference/mean": 0.08512242883443832, "step": 638 }, { "clip_ratio/high_max": 0.026776819955557585, "clip_ratio/high_mean": 0.026776819955557585, "clip_ratio/low_mean": 0.015207641525194049, "clip_ratio/low_min": 0.015207641525194049, "clip_ratio/region_mean": 0.041984461480751634, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 74.75, "completions/mean_terminated_length": 74.75, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.3610265199095011, "epoch": 0.024675625579240038, "frac_reward_zero_std": 0.0, "grad_norm": 5.633913993835449, "learning_rate": 8.066666666666667e-06, "loss": -0.0012, "num_tokens": 1383384.0, "reward": 0.8388459086418152, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8388459086418152, "reward_meter_std": 0.19675958156585693, "reward_std": 0.19675958156585693, "reward_total_composite_mean": 0.8388459086418152, "reward_total_composite_std": 0.19675958156585693, "reward_total_mean": 0.8388459086418152, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8388459086418152, "rewards/meter/std": 0.19675958156585693, "rewards/total_composite/mean": 0.8388459086418152, "rewards/total_composite/std": 0.19675958156585693, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.998264491558075, "sampling/importance_sampling_ratio/min": 0.21954980492591858, "sampling/sampling_logp_difference/max": 1.5161762237548828, "sampling/sampling_logp_difference/mean": 0.05180966109037399, "step": 639 }, { "clip_ratio/high_max": 0.0054954917868599296, "clip_ratio/high_mean": 0.0054954917868599296, "clip_ratio/low_mean": 0.002071823226287961, "clip_ratio/low_min": 0.002071823226287961, "clip_ratio/region_mean": 0.007567315013147891, "completions/clipped_ratio": 0.0, "completions/max_length": 181.0, "completions/max_terminated_length": 181.0, "completions/mean_length": 162.25, "completions/mean_terminated_length": 162.25, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.01887570507824421, "epoch": 0.02471424158171146, "frac_reward_zero_std": 0.0, "grad_norm": 0.2766904830932617, "learning_rate": 8.063636363636364e-06, "loss": 0.0423, "num_tokens": 1386090.0, "reward": 0.970228910446167, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.9950998425483704, "reward_meter_std": 0.0002485640870872885, "reward_std": 0.0704522356390953, "reward_total_composite_mean": 0.970228910446167, "reward_total_composite_std": 0.0704522356390953, "reward_total_mean": 0.970228910446167, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.9950998425483704, "rewards/meter/std": 0.0002485640870872885, "rewards/total_composite/mean": 0.970228910446167, "rewards/total_composite/std": 0.0704522356390953, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0019837617874146, "sampling/importance_sampling_ratio/min": 0.07809732854366302, "sampling/sampling_logp_difference/max": 2.5497994422912598, "sampling/sampling_logp_difference/mean": 0.007383763324469328, "step": 640 }, { "clip_ratio/high_max": 0.025679388898424804, "clip_ratio/high_mean": 0.025679388898424804, "clip_ratio/low_mean": 0.012743587838485837, "clip_ratio/low_min": 0.012743587838485837, "clip_ratio/region_mean": 0.03842297673691064, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 76.75, "completions/mean_terminated_length": 76.75, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.1969589926302433, "epoch": 0.024752857584182886, "frac_reward_zero_std": 0.0, "grad_norm": 3.8018136024475098, "learning_rate": 8.060606060606061e-06, "loss": -0.0001, "num_tokens": 1387880.0, "reward": 0.9926354885101318, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9926354885101318, "reward_meter_std": 0.003230726346373558, "reward_std": 0.0032307212240993977, "reward_total_composite_mean": 0.9926354885101318, "reward_total_composite_std": 0.003230726346373558, "reward_total_mean": 0.9926354885101318, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9926354885101318, "rewards/meter/std": 0.003230726346373558, "rewards/total_composite/mean": 0.9926354885101318, "rewards/total_composite/std": 0.003230726346373558, "sampling/importance_sampling_ratio/max": 1.8937081098556519, "sampling/importance_sampling_ratio/mean": 1.0071512460708618, "sampling/importance_sampling_ratio/min": 0.37865278124809265, "sampling/sampling_logp_difference/max": 0.9711356163024902, "sampling/sampling_logp_difference/mean": 0.03708671033382416, "step": 641 }, { "clip_ratio/high_max": 0.053032459458336234, "clip_ratio/high_mean": 0.053032459458336234, "clip_ratio/low_mean": 0.012456294149160385, "clip_ratio/low_min": 0.012456294149160385, "clip_ratio/region_mean": 0.06548875360749662, "completions/clipped_ratio": 0.0, "completions/max_length": 47.0, "completions/max_terminated_length": 47.0, "completions/mean_length": 40.25, "completions/mean_terminated_length": 40.25, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.34129922464489937, "epoch": 0.02479147358665431, "frac_reward_zero_std": 0.0, "grad_norm": 10.998772621154785, "learning_rate": 8.057575757575759e-06, "loss": 0.0512, "num_tokens": 1389306.0, "reward": 0.9872698783874512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9872698783874512, "reward_meter_std": 0.01214898843318224, "reward_std": 0.012148984707891941, "reward_total_composite_mean": 0.9872698783874512, "reward_total_composite_std": 0.01214898843318224, "reward_total_mean": 0.9872698783874512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9872698783874512, "rewards/meter/std": 0.01214898843318224, "rewards/total_composite/mean": 0.9872698783874512, "rewards/total_composite/std": 0.01214898843318224, "sampling/importance_sampling_ratio/max": 1.8077318668365479, "sampling/importance_sampling_ratio/mean": 1.0039377212524414, "sampling/importance_sampling_ratio/min": 0.27446630597114563, "sampling/sampling_logp_difference/max": 1.2929267883300781, "sampling/sampling_logp_difference/mean": 0.0713094025850296, "step": 642 }, { "clip_ratio/high_max": 0.0059894436853937805, "clip_ratio/high_mean": 0.0059894436853937805, "clip_ratio/low_mean": 0.011697308160364628, "clip_ratio/low_min": 0.011697308160364628, "clip_ratio/region_mean": 0.01768675184575841, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 246.375, "completions/mean_terminated_length": 246.375, "completions/min_length": 192.0, "completions/min_terminated_length": 192.0, "entropy": 0.11560235964134336, "epoch": 0.024830089589125734, "frac_reward_zero_std": 0.0, "grad_norm": 3.1086318492889404, "learning_rate": 8.054545454545454e-06, "loss": 0.11, "num_tokens": 1392997.0, "reward": 0.3722308278083801, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.930555522441864, "reward_count_adherence_std": 0.05750546231865883, "reward_meter_mean": 0.3895261585712433, "reward_meter_std": 0.42622098326683044, "reward_std": 0.41810813546180725, "reward_total_composite_mean": 0.3722308278083801, "reward_total_composite_std": 0.41810813546180725, "reward_total_mean": 0.3722308278083801, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.930555522441864, "rewards/count_adherence/std": 0.05750546231865883, "rewards/meter/mean": 0.3895261585712433, "rewards/meter/std": 0.42622098326683044, "rewards/total_composite/mean": 0.3722308278083801, "rewards/total_composite/std": 0.41810813546180725, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0075058937072754, "sampling/importance_sampling_ratio/min": 0.046687766909599304, "sampling/sampling_logp_difference/max": 3.0642731189727783, "sampling/sampling_logp_difference/mean": 0.02562355063855648, "step": 643 }, { "clip_ratio/high_max": 0.0217176906298846, "clip_ratio/high_mean": 0.0217176906298846, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0217176906298846, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 114.0, "completions/mean_terminated_length": 57.142860412597656, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.13272713590413332, "epoch": 0.024868705591597158, "frac_reward_zero_std": 0.0, "grad_norm": 0.7139359712600708, "learning_rate": 8.051515151515153e-06, "loss": -0.1547, "num_tokens": 1394733.0, "reward": 0.8692982196807861, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8798133134841919, "reward_meter_std": 0.321513295173645, "reward_std": 0.3512541353702545, "reward_total_composite_mean": 0.8692982196807861, "reward_total_composite_std": 0.3512541353702545, "reward_total_mean": 0.8692982196807861, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8798133134841919, "rewards/meter/std": 0.321513295173645, "rewards/total_composite/mean": 0.8692982196807861, "rewards/total_composite/std": 0.3512541353702545, "sampling/importance_sampling_ratio/max": 1.4886727333068848, "sampling/importance_sampling_ratio/mean": 0.9990324378013611, "sampling/importance_sampling_ratio/min": 0.2924591600894928, "sampling/sampling_logp_difference/max": 1.2294301986694336, "sampling/sampling_logp_difference/mean": 0.0238660741597414, "step": 644 }, { "clip_ratio/high_max": 0.008699847152456641, "clip_ratio/high_mean": 0.008699847152456641, "clip_ratio/low_mean": 0.007154436782002449, "clip_ratio/low_min": 0.007154436782002449, "clip_ratio/region_mean": 0.01585428393445909, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 127.25, "completions/mean_terminated_length": 127.25, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.07990281283855438, "epoch": 0.024907321594068582, "frac_reward_zero_std": 0.0, "grad_norm": 2.8207523822784424, "learning_rate": 8.048484848484849e-06, "loss": -0.0075, "num_tokens": 1397215.0, "reward": 0.5935784578323364, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5935784578323364, "reward_meter_std": 0.49478766322135925, "reward_std": 0.49478766322135925, "reward_total_composite_mean": 0.5935784578323364, "reward_total_composite_std": 0.49478766322135925, "reward_total_mean": 0.5935784578323364, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5935784578323364, "rewards/meter/std": 0.49478766322135925, "rewards/total_composite/mean": 0.5935784578323364, "rewards/total_composite/std": 0.49478766322135925, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0039057731628418, "sampling/importance_sampling_ratio/min": 0.18896886706352234, "sampling/sampling_logp_difference/max": 1.666172981262207, "sampling/sampling_logp_difference/mean": 0.0225541889667511, "step": 645 }, { "clip_ratio/high_max": 0.012931034667417407, "clip_ratio/high_mean": 0.012931034667417407, "clip_ratio/low_mean": 0.015636918134987354, "clip_ratio/low_min": 0.015636918134987354, "clip_ratio/region_mean": 0.02856795280240476, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 172.75, "completions/mean_terminated_length": 59.66666793823242, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.23534639924764633, "epoch": 0.024945937596540006, "frac_reward_zero_std": 0.0, "grad_norm": 1.6291898488998413, "learning_rate": 8.045454545454546e-06, "loss": -0.0732, "num_tokens": 1399037.0, "reward": 0.49611878395080566, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.5550702810287476, "reward_meter_std": 0.4581260085105896, "reward_std": 0.49892619252204895, "reward_total_composite_mean": 0.49611878395080566, "reward_total_composite_std": 0.49892622232437134, "reward_total_mean": 0.49611878395080566, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.5550702810287476, "rewards/meter/std": 0.4581260085105896, "rewards/total_composite/mean": 0.49611878395080566, "rewards/total_composite/std": 0.49892622232437134, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0126827955245972, "sampling/importance_sampling_ratio/min": 0.22186465561389923, "sampling/sampling_logp_difference/max": 1.5663788318634033, "sampling/sampling_logp_difference/mean": 0.05649138242006302, "step": 646 }, { "clip_ratio/high_max": 0.05566767114214599, "clip_ratio/high_mean": 0.05566767114214599, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.05753334274049848, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.2588699162006378, "epoch": 0.02498455359901143, "frac_reward_zero_std": 0.0, "grad_norm": 10.47767162322998, "learning_rate": 8.042424242424243e-06, "loss": 0.0108, "num_tokens": 1400768.0, "reward": 0.9769783020019531, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9769783020019531, "reward_meter_std": 0.051688529551029205, "reward_std": 0.05168852210044861, "reward_total_composite_mean": 0.9769783020019531, "reward_total_composite_std": 0.051688529551029205, "reward_total_mean": 0.9769783020019531, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9769783020019531, "rewards/meter/std": 0.051688529551029205, "rewards/total_composite/mean": 0.9769783020019531, "rewards/total_composite/std": 0.051688529551029205, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9968314170837402, "sampling/importance_sampling_ratio/min": 0.20647425949573517, "sampling/sampling_logp_difference/max": 1.5775794982910156, "sampling/sampling_logp_difference/mean": 0.056386951357126236, "step": 647 }, { "clip_ratio/high_max": 0.009631438297219574, "clip_ratio/high_mean": 0.009631438297219574, "clip_ratio/low_mean": 0.005490340990945697, "clip_ratio/low_min": 0.005490340990945697, "clip_ratio/region_mean": 0.015121779288165271, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 250.375, "completions/mean_terminated_length": 250.375, "completions/min_length": 222.0, "completions/min_terminated_length": 222.0, "entropy": 0.06133082904852927, "epoch": 0.025023169601482854, "frac_reward_zero_std": 0.0, "grad_norm": 2.8141632080078125, "learning_rate": 8.03939393939394e-06, "loss": 0.0133, "num_tokens": 1404355.0, "reward": 0.6260231733322144, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_meter_mean": 0.7232567071914673, "reward_meter_std": 0.3504089117050171, "reward_std": 0.3030511140823364, "reward_total_composite_mean": 0.6260231733322144, "reward_total_composite_std": 0.3030511140823364, "reward_total_mean": 0.6260231733322144, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/meter/mean": 0.7232567071914673, "rewards/meter/std": 0.3504089117050171, "rewards/total_composite/mean": 0.6260231733322144, "rewards/total_composite/std": 0.3030511140823364, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9957653880119324, "sampling/importance_sampling_ratio/min": 0.004950075875967741, "sampling/sampling_logp_difference/max": 5.308352470397949, "sampling/sampling_logp_difference/mean": 0.026500631123781204, "step": 648 }, { "clip_ratio/high_max": 0.011183355352841318, "clip_ratio/high_mean": 0.011183355352841318, "clip_ratio/low_mean": 0.006225490127690136, "clip_ratio/low_min": 0.006225490127690136, "clip_ratio/region_mean": 0.017408845480531454, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 112.25, "completions/mean_terminated_length": 112.25, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.11915135849267244, "epoch": 0.02506178560395428, "frac_reward_zero_std": 0.0, "grad_norm": 3.6201226711273193, "learning_rate": 8.036363636363636e-06, "loss": 0.0372, "num_tokens": 1406589.0, "reward": 0.981624960899353, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.981624960899353, "reward_meter_std": 0.027504365891218185, "reward_std": 0.02750437520444393, "reward_total_composite_mean": 0.981624960899353, "reward_total_composite_std": 0.027504365891218185, "reward_total_mean": 0.981624960899353, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.981624960899353, "rewards/meter/std": 0.027504365891218185, "rewards/total_composite/mean": 0.981624960899353, "rewards/total_composite/std": 0.027504365891218185, "sampling/importance_sampling_ratio/max": 1.9342072010040283, "sampling/importance_sampling_ratio/mean": 1.0045708417892456, "sampling/importance_sampling_ratio/min": 0.3109181523323059, "sampling/sampling_logp_difference/max": 1.1682255268096924, "sampling/sampling_logp_difference/mean": 0.019951220601797104, "step": 649 }, { "clip_ratio/high_max": 0.05512813315726817, "clip_ratio/high_mean": 0.05512813315726817, "clip_ratio/low_mean": 0.033882785588502884, "clip_ratio/low_min": 0.033882785588502884, "clip_ratio/region_mean": 0.08901091874577105, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 35.625, "completions/mean_terminated_length": 35.625, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.5731233917176723, "epoch": 0.025100401606425703, "frac_reward_zero_std": 0.0, "grad_norm": 12.612115859985352, "learning_rate": 8.033333333333335e-06, "loss": 0.0674, "num_tokens": 1408050.0, "reward": 0.9578008651733398, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9578008651733398, "reward_meter_std": 0.05439894273877144, "reward_std": 0.05439893528819084, "reward_total_composite_mean": 0.9578008651733398, "reward_total_composite_std": 0.05439894273877144, "reward_total_mean": 0.9578008651733398, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9578008651733398, "rewards/meter/std": 0.05439894273877144, "rewards/total_composite/mean": 0.9578008651733398, "rewards/total_composite/std": 0.05439894273877144, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007057547569275, "sampling/importance_sampling_ratio/min": 0.18632878363132477, "sampling/sampling_logp_difference/max": 1.6802425384521484, "sampling/sampling_logp_difference/mean": 0.11987494677305222, "step": 650 }, { "epoch": 0.025100401606425703, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/max_length": 433.15384615384613, "eval_completions/max_terminated_length": 416.6923076923077, "eval_completions/mean_length": 229.43269230769232, "eval_completions/mean_terminated_length": 220.6758258526142, "eval_completions/min_length": 59.38461538461539, "eval_completions/min_terminated_length": 59.38461538461539, "eval_entropy": 0.07029147646748103, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 1408050.0, "eval_reward": 0.7413886831356928, "eval_reward_arabic_clean_mean": 0.9615384615384616, "eval_reward_arabic_clean_std": 0.0900012942460867, "eval_reward_count_adherence_mean": 0.9108192599736727, "eval_reward_count_adherence_std": 0.1330794313779244, "eval_reward_meter_mean": 0.8344126389576838, "eval_reward_meter_std": 0.3256990680327782, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7413886831356928, "eval_reward_total_composite_std": 0.3471943082717749, "eval_reward_total_mean": 0.7413886831356928, "eval_rewards/arabic_clean/mean": 0.9615384615384616, "eval_rewards/arabic_clean/std": 0.0900012942460867, "eval_rewards/count_adherence/mean": 0.9108192599736727, "eval_rewards/count_adherence/std": 0.1330794313779244, "eval_rewards/meter/mean": 0.8344126389576838, "eval_rewards/meter/std": 0.3256990680327782, "eval_rewards/total_composite/mean": 0.7413886831356928, "eval_rewards/total_composite/std": 0.3471943082717749, "eval_runtime": 81.5104, "eval_samples_per_second": 1.276, "eval_sampling/importance_sampling_ratio/max": 1.3485183532421405, "eval_sampling/importance_sampling_ratio/mean": 1.0019580951103797, "eval_sampling/importance_sampling_ratio/min": 0.3777222243639139, "eval_sampling/sampling_logp_difference/max": 0.9992800859304575, "eval_sampling/sampling_logp_difference/mean": 0.007470924049042738, "eval_steps_per_second": 0.159, "step": 650 }, { "clip_ratio/high_max": 0.010521235642954707, "clip_ratio/high_mean": 0.010521235642954707, "clip_ratio/low_mean": 0.028315248200669885, "clip_ratio/low_min": 0.028315248200669885, "clip_ratio/region_mean": 0.03883648384362459, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 35.125, "completions/mean_terminated_length": 35.125, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.17123969458043575, "epoch": 0.026147728642005062, "frac_reward_zero_std": 0.0, "grad_norm": 9.100071907043457, "learning_rate": 8.03030303030303e-06, "loss": -0.0071, "num_tokens": 1409555.0, "reward": 0.9935092329978943, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9935092329978943, "reward_meter_std": 0.000985646271146834, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009856420801952481, "reward_total_composite_mean": 0.9935092329978943, "reward_total_composite_std": 0.000985646271146834, "reward_total_mean": 0.9935092329978943, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9935092329978943, "rewards/meter/std": 0.000985646271146834, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9935092329978943, "rewards/total_composite/std": 0.000985646271146834, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.011391282081604, "sampling/importance_sampling_ratio/min": 0.3883599042892456, "sampling/sampling_logp_difference/max": 0.9458228349685669, "sampling/sampling_logp_difference/mean": 0.04576735198497772, "step": 651 }, { "clip_ratio/high_max": 0.0035446555411908776, "clip_ratio/high_mean": 0.0035446555411908776, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0035446555411908776, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 363.0, "completions/mean_length": 374.25, "completions/mean_terminated_length": 354.5714416503906, "completions/min_length": 343.0, "completions/min_terminated_length": 343.0, "entropy": 0.019337893230840564, "epoch": 0.026187894123790016, "frac_reward_zero_std": 0.0, "grad_norm": 0.4141400456428528, "learning_rate": 8.027272727272728e-06, "loss": -0.2894, "num_tokens": 1413749.0, "reward": 0.04464861750602722, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.337003618478775, "reward_meter_mean": 0.8732885122299194, "reward_meter_std": 0.3528631627559662, "reward_repeat_penalty_mean": 0.17251461744308472, "reward_repeat_penalty_std": 0.33435773849487305, "reward_std": 0.018085261806845665, "reward_total_composite_mean": 0.04464861750602722, "reward_total_composite_std": 0.018085261806845665, "reward_total_mean": 0.04464861750602722, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.337003618478775, "rewards/meter/mean": 0.8732885122299194, "rewards/meter/std": 0.3528631627559662, "rewards/repeat_penalty/mean": 0.17251461744308472, "rewards/repeat_penalty/std": 0.33435773849487305, "rewards/total_composite/mean": 0.04464861750602722, "rewards/total_composite/std": 0.018085261806845665, "sampling/importance_sampling_ratio/max": 1.834702491760254, "sampling/importance_sampling_ratio/mean": 1.0013567209243774, "sampling/importance_sampling_ratio/min": 0.4920751452445984, "sampling/sampling_logp_difference/max": 0.7091238498687744, "sampling/sampling_logp_difference/mean": 0.004473666660487652, "step": 652 }, { "clip_ratio/high_max": 0.0029761905316263437, "clip_ratio/high_mean": 0.0029761905316263437, "clip_ratio/low_mean": 0.007873826543800533, "clip_ratio/low_min": 0.007873826543800533, "clip_ratio/region_mean": 0.010850017075426877, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 133.5, "completions/mean_terminated_length": 79.42857360839844, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.0790142323821783, "epoch": 0.02622805960557497, "frac_reward_zero_std": 0.0, "grad_norm": 1.191830039024353, "learning_rate": 8.024242424242425e-06, "loss": 0.1416, "num_tokens": 1415617.0, "reward": 0.43250638246536255, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9687988758087158, "reward_meter_std": 0.06074121594429016, "reward_repeat_penalty_mean": 0.4583333432674408, "reward_repeat_penalty_std": 0.24800792336463928, "reward_std": 0.1944577544927597, "reward_total_composite_mean": 0.43250638246536255, "reward_total_composite_std": 0.1944577693939209, "reward_total_mean": 0.43250638246536255, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9687988758087158, "rewards/meter/std": 0.06074121594429016, "rewards/repeat_penalty/mean": 0.4583333432674408, "rewards/repeat_penalty/std": 0.24800792336463928, "rewards/total_composite/mean": 0.43250638246536255, "rewards/total_composite/std": 0.1944577693939209, "sampling/importance_sampling_ratio/max": 1.5146371126174927, "sampling/importance_sampling_ratio/mean": 1.0012526512145996, "sampling/importance_sampling_ratio/min": 0.3877831697463989, "sampling/sampling_logp_difference/max": 0.9473090171813965, "sampling/sampling_logp_difference/mean": 0.014217361807823181, "step": 653 }, { "clip_ratio/high_max": 0.008620842476375401, "clip_ratio/high_mean": 0.008620842476375401, "clip_ratio/low_mean": 0.003801907878369093, "clip_ratio/low_min": 0.003801907878369093, "clip_ratio/region_mean": 0.012422750354744494, "completions/clipped_ratio": 0.0, "completions/max_length": 183.0, "completions/max_terminated_length": 183.0, "completions/mean_length": 170.625, "completions/mean_terminated_length": 170.625, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.08471588138490915, "epoch": 0.026268225087359924, "frac_reward_zero_std": 0.0, "grad_norm": 1.927111268043518, "learning_rate": 8.021212121212122e-06, "loss": -0.0232, "num_tokens": 1418478.0, "reward": 0.2665513753890991, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9953961968421936, "reward_meter_std": 0.0023201238363981247, "reward_repeat_penalty_mean": 0.2678571343421936, "reward_repeat_penalty_std": 0.17806050181388855, "reward_std": 0.17725922167301178, "reward_total_composite_mean": 0.2665513753890991, "reward_total_composite_std": 0.17725923657417297, "reward_total_mean": 0.2665513753890991, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9953961968421936, "rewards/meter/std": 0.0023201238363981247, "rewards/repeat_penalty/mean": 0.2678571343421936, "rewards/repeat_penalty/std": 0.17806050181388855, "rewards/total_composite/mean": 0.2665513753890991, "rewards/total_composite/std": 0.17725923657417297, "sampling/importance_sampling_ratio/max": 1.6224781274795532, "sampling/importance_sampling_ratio/mean": 1.0009305477142334, "sampling/importance_sampling_ratio/min": 0.24168278276920319, "sampling/sampling_logp_difference/max": 1.4201292991638184, "sampling/sampling_logp_difference/mean": 0.01804671622812748, "step": 654 }, { "clip_ratio/high_max": 0.012714299838989973, "clip_ratio/high_mean": 0.012714299838989973, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.01455253513995558, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 116.625, "completions/mean_terminated_length": 60.142860412597656, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.17866009753197432, "epoch": 0.026308390569144878, "frac_reward_zero_std": 0.0, "grad_norm": 2.2248940467834473, "learning_rate": 8.018181818181818e-06, "loss": -0.0945, "num_tokens": 1420163.0, "reward": 0.25364238023757935, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7503823637962341, "reward_meter_std": 0.44998785853385925, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.30860671401023865, "reward_std": 0.1438898742198944, "reward_total_composite_mean": 0.25364238023757935, "reward_total_composite_std": 0.14388985931873322, "reward_total_mean": 0.25364238023757935, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7503823637962341, "rewards/meter/std": 0.44998785853385925, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.30860671401023865, "rewards/total_composite/mean": 0.25364238023757935, "rewards/total_composite/std": 0.14388985931873322, "sampling/importance_sampling_ratio/max": 1.6566717624664307, "sampling/importance_sampling_ratio/mean": 1.0044450759887695, "sampling/importance_sampling_ratio/min": 0.35307735204696655, "sampling/sampling_logp_difference/max": 1.0410680770874023, "sampling/sampling_logp_difference/mean": 0.024109482765197754, "step": 655 }, { "clip_ratio/high_max": 0.01278492109850049, "clip_ratio/high_mean": 0.01278492109850049, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/region_mean": 0.01709526591002941, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.16807558294385672, "epoch": 0.026348556050929832, "frac_reward_zero_std": 0.0, "grad_norm": 6.352079391479492, "learning_rate": 8.015151515151515e-06, "loss": 0.0094, "num_tokens": 1421891.0, "reward": 0.3079977035522461, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9239930510520935, "reward_meter_std": 0.15441425144672394, "reward_repeat_penalty_mean": 0.3333333432674408, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05147142335772514, "reward_total_composite_mean": 0.3079977035522461, "reward_total_composite_std": 0.051471415907144547, "reward_total_mean": 0.3079977035522461, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9239930510520935, "rewards/meter/std": 0.15441425144672394, "rewards/repeat_penalty/mean": 0.3333333432674408, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3079977035522461, "rewards/total_composite/std": 0.051471415907144547, "sampling/importance_sampling_ratio/max": 1.928532361984253, "sampling/importance_sampling_ratio/mean": 1.0036613941192627, "sampling/importance_sampling_ratio/min": 0.41602474451065063, "sampling/sampling_logp_difference/max": 0.8770105838775635, "sampling/sampling_logp_difference/mean": 0.022419268265366554, "step": 656 }, { "clip_ratio/high_max": 0.0028169237775728106, "clip_ratio/high_mean": 0.0028169237775728106, "clip_ratio/low_mean": 0.008925562433432788, "clip_ratio/low_min": 0.008925562433432788, "clip_ratio/region_mean": 0.011742486211005598, "completions/clipped_ratio": 0.0, "completions/max_length": 185.0, "completions/max_terminated_length": 185.0, "completions/mean_length": 181.25, "completions/mean_terminated_length": 181.25, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.042974324664101005, "epoch": 0.026388721532714786, "frac_reward_zero_std": 0.0, "grad_norm": 2.9878616333007812, "learning_rate": 8.012121212121214e-06, "loss": 0.0154, "num_tokens": 1424909.0, "reward": 0.1662987768650055, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9874005317687988, "reward_meter_std": 0.011436098255217075, "reward_repeat_penalty_mean": 0.20192307233810425, "reward_repeat_penalty_std": 0.2094048410654068, "reward_std": 0.17312301695346832, "reward_total_composite_mean": 0.1662987768650055, "reward_total_composite_std": 0.17312301695346832, "reward_total_mean": 0.1662987768650055, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9874005317687988, "rewards/meter/std": 0.011436098255217075, "rewards/repeat_penalty/mean": 0.20192307233810425, "rewards/repeat_penalty/std": 0.2094048410654068, "rewards/total_composite/mean": 0.1662987768650055, "rewards/total_composite/std": 0.17312301695346832, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016026496887207, "sampling/importance_sampling_ratio/min": 0.1793055236339569, "sampling/sampling_logp_difference/max": 1.7186641693115234, "sampling/sampling_logp_difference/mean": 0.010927603580057621, "step": 657 }, { "clip_ratio/high_max": 0.0005817760829813778, "clip_ratio/high_mean": 0.0005817760829813778, "clip_ratio/low_mean": 0.00030637255986221135, "clip_ratio/low_min": 0.00030637255986221135, "clip_ratio/region_mean": 0.0008881486428435892, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 441.0, "completions/mean_length": 428.5, "completions/mean_terminated_length": 416.5714416503906, "completions/min_length": 394.0, "completions/min_terminated_length": 394.0, "entropy": 0.012874894309788942, "epoch": 0.02642888701449974, "frac_reward_zero_std": 0.0, "grad_norm": 0.7183297276496887, "learning_rate": 8.00909090909091e-06, "loss": -0.0762, "num_tokens": 1429785.0, "reward": 0.033445145934820175, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.6416666507720947, "reward_count_adherence_std": 0.26170989871025085, "reward_meter_mean": 0.8735865354537964, "reward_meter_std": 0.35298284888267517, "reward_repeat_penalty_mean": 0.17413419485092163, "reward_repeat_penalty_std": 0.34882497787475586, "reward_std": 0.06880706548690796, "reward_total_composite_mean": 0.033445145934820175, "reward_total_composite_std": 0.06880706548690796, "reward_total_mean": 0.033445145934820175, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.6416666507720947, "rewards/count_adherence/std": 0.26170989871025085, "rewards/meter/mean": 0.8735865354537964, "rewards/meter/std": 0.35298284888267517, "rewards/repeat_penalty/mean": 0.17413419485092163, "rewards/repeat_penalty/std": 0.34882497787475586, "rewards/total_composite/mean": 0.033445145934820175, "rewards/total_composite/std": 0.06880706548690796, "sampling/importance_sampling_ratio/max": 1.7735869884490967, "sampling/importance_sampling_ratio/mean": 1.0009920597076416, "sampling/importance_sampling_ratio/min": 0.3645470440387726, "sampling/sampling_logp_difference/max": 1.0090997219085693, "sampling/sampling_logp_difference/mean": 0.0026275559794157743, "step": 658 }, { "clip_ratio/high_max": 0.008969587041065097, "clip_ratio/high_mean": 0.008969587041065097, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.008969587041065097, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 190.375, "completions/mean_terminated_length": 83.16667175292969, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.06744233565405011, "epoch": 0.026469052496284694, "frac_reward_zero_std": 0.0, "grad_norm": 1.3609012365341187, "learning_rate": 8.006060606060607e-06, "loss": -0.1652, "num_tokens": 1431540.0, "reward": 0.513268232345581, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9505882263183594, "reward_meter_std": 0.030698692426085472, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.3378343880176544, "reward_total_composite_mean": 0.513268232345581, "reward_total_composite_std": 0.3378344178199768, "reward_total_mean": 0.513268232345581, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9505882263183594, "rewards/meter/std": 0.030698692426085472, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.513268232345581, "rewards/total_composite/std": 0.3378344178199768, "sampling/importance_sampling_ratio/max": 1.7507625818252563, "sampling/importance_sampling_ratio/mean": 1.0079931020736694, "sampling/importance_sampling_ratio/min": 0.1503116339445114, "sampling/sampling_logp_difference/max": 1.8950445652008057, "sampling/sampling_logp_difference/mean": 0.02022649347782135, "step": 659 }, { "clip_ratio/high_max": 0.0036057692486792803, "clip_ratio/high_mean": 0.0036057692486792803, "clip_ratio/low_mean": 0.006200430449098349, "clip_ratio/low_min": 0.006200430449098349, "clip_ratio/region_mean": 0.009806199697777629, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 110.625, "completions/mean_terminated_length": 110.625, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.020129066659137607, "epoch": 0.026509217978069648, "frac_reward_zero_std": 0.0, "grad_norm": 2.2579872608184814, "learning_rate": 8.003030303030304e-06, "loss": 0.0308, "num_tokens": 1433681.0, "reward": 0.22219520807266235, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.9927444458007812, "reward_meter_std": 0.0012347318697720766, "reward_repeat_penalty_mean": 0.23571428656578064, "reward_repeat_penalty_std": 0.07284314185380936, "reward_std": 0.07082199305295944, "reward_total_composite_mean": 0.22219520807266235, "reward_total_composite_std": 0.07082200050354004, "reward_total_mean": 0.22219520807266235, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.9927444458007812, "rewards/meter/std": 0.0012347318697720766, "rewards/repeat_penalty/mean": 0.23571428656578064, "rewards/repeat_penalty/std": 0.07284314185380936, "rewards/total_composite/mean": 0.22219520807266235, "rewards/total_composite/std": 0.07082200050354004, "sampling/importance_sampling_ratio/max": 1.6747108697891235, "sampling/importance_sampling_ratio/mean": 1.001028299331665, "sampling/importance_sampling_ratio/min": 0.1580846756696701, "sampling/sampling_logp_difference/max": 1.8446245193481445, "sampling/sampling_logp_difference/mean": 0.008687403053045273, "step": 660 }, { "clip_ratio/high_max": 0.017718179151415825, "clip_ratio/high_mean": 0.017718179151415825, "clip_ratio/low_mean": 0.008919409476220608, "clip_ratio/low_min": 0.008919409476220608, "clip_ratio/region_mean": 0.026637588627636433, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 203.625, "completions/mean_terminated_length": 100.83333587646484, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.17050567921251059, "epoch": 0.0265493834598546, "frac_reward_zero_std": 0.0, "grad_norm": 3.181408405303955, "learning_rate": 8.000000000000001e-06, "loss": -0.0515, "num_tokens": 1435750.0, "reward": 0.22007080912590027, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.18600596487522125, "reward_meter_mean": 0.43306416273117065, "reward_meter_std": 0.44791871309280396, "reward_repeat_penalty_mean": 0.6357142925262451, "reward_repeat_penalty_std": 0.31916436553001404, "reward_std": 0.2968771457672119, "reward_total_composite_mean": 0.22007080912590027, "reward_total_composite_std": 0.2968771755695343, "reward_total_mean": 0.22007080912590027, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.18600596487522125, "rewards/meter/mean": 0.43306416273117065, "rewards/meter/std": 0.44791871309280396, "rewards/repeat_penalty/mean": 0.6357142925262451, "rewards/repeat_penalty/std": 0.31916436553001404, "rewards/total_composite/mean": 0.22007080912590027, "rewards/total_composite/std": 0.2968771755695343, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0068386793136597, "sampling/importance_sampling_ratio/min": 0.253473699092865, "sampling/sampling_logp_difference/max": 1.372495174407959, "sampling/sampling_logp_difference/mean": 0.04375706985592842, "step": 661 }, { "clip_ratio/high_max": 0.012073024990968406, "clip_ratio/high_mean": 0.012073024990968406, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/region_mean": 0.016458989935927093, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 117.0, "completions/mean_terminated_length": 60.57143020629883, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.17267457023262978, "epoch": 0.026589548941639556, "frac_reward_zero_std": 0.0, "grad_norm": 2.6338069438934326, "learning_rate": 7.996969696969697e-06, "loss": -0.1279, "num_tokens": 1437638.0, "reward": 0.455108642578125, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9266777634620667, "reward_meter_std": 0.1883465051651001, "reward_repeat_penalty_mean": 0.5833333730697632, "reward_repeat_penalty_std": 0.2357022762298584, "reward_std": 0.24599777162075043, "reward_total_composite_mean": 0.455108642578125, "reward_total_composite_std": 0.24599777162075043, "reward_total_mean": 0.455108642578125, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9266777634620667, "rewards/meter/std": 0.1883465051651001, "rewards/repeat_penalty/mean": 0.5833333730697632, "rewards/repeat_penalty/std": 0.2357022762298584, "rewards/total_composite/mean": 0.455108642578125, "rewards/total_composite/std": 0.24599777162075043, "sampling/importance_sampling_ratio/max": 1.474334478378296, "sampling/importance_sampling_ratio/mean": 1.0021228790283203, "sampling/importance_sampling_ratio/min": 0.263834148645401, "sampling/sampling_logp_difference/max": 1.3324346542358398, "sampling/sampling_logp_difference/mean": 0.032812319695949554, "step": 662 }, { "clip_ratio/high_max": 0.014703674940392375, "clip_ratio/high_mean": 0.014703674940392375, "clip_ratio/low_mean": 0.00461137923412025, "clip_ratio/low_min": 0.00461137923412025, "clip_ratio/region_mean": 0.019315054174512625, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 108.0, "completions/mean_terminated_length": 108.0, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.11175651382654905, "epoch": 0.02662971442342451, "frac_reward_zero_std": 0.0, "grad_norm": 4.197593688964844, "learning_rate": 7.993939393939396e-06, "loss": 0.021, "num_tokens": 1439950.0, "reward": 0.3900730013847351, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9684537053108215, "reward_meter_std": 0.027583837509155273, "reward_repeat_penalty_mean": 0.4226190447807312, "reward_repeat_penalty_std": 0.2210753709077835, "reward_std": 0.19891956448554993, "reward_total_composite_mean": 0.3900730013847351, "reward_total_composite_std": 0.19891957938671112, "reward_total_mean": 0.3900730013847351, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9684537053108215, "rewards/meter/std": 0.027583837509155273, "rewards/repeat_penalty/mean": 0.4226190447807312, "rewards/repeat_penalty/std": 0.2210753709077835, "rewards/total_composite/mean": 0.3900730013847351, "rewards/total_composite/std": 0.19891957938671112, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9997091889381409, "sampling/importance_sampling_ratio/min": 0.3571523129940033, "sampling/sampling_logp_difference/max": 1.0295929908752441, "sampling/sampling_logp_difference/mean": 0.02085985243320465, "step": 663 }, { "clip_ratio/high_max": 0.023867564275860786, "clip_ratio/high_mean": 0.023867564275860786, "clip_ratio/low_mean": 0.0055147059028968215, "clip_ratio/low_min": 0.0055147059028968215, "clip_ratio/region_mean": 0.029382270178757608, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 70.625, "completions/mean_terminated_length": 70.625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.10323597816750407, "epoch": 0.026669879905209463, "frac_reward_zero_std": 0.0, "grad_norm": 7.99793004989624, "learning_rate": 7.990909090909091e-06, "loss": -0.029, "num_tokens": 1441763.0, "reward": 0.568142294883728, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9828510880470276, "reward_meter_std": 0.025825461372733116, "reward_repeat_penalty_mean": 0.5833333730697632, "reward_repeat_penalty_std": 0.29546841979026794, "reward_std": 0.27455058693885803, "reward_total_composite_mean": 0.568142294883728, "reward_total_composite_std": 0.27455055713653564, "reward_total_mean": 0.568142294883728, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9828510880470276, "rewards/meter/std": 0.025825461372733116, "rewards/repeat_penalty/mean": 0.5833333730697632, "rewards/repeat_penalty/std": 0.29546841979026794, "rewards/total_composite/mean": 0.568142294883728, "rewards/total_composite/std": 0.27455055713653564, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028775930404663, "sampling/importance_sampling_ratio/min": 0.3252873122692108, "sampling/sampling_logp_difference/max": 1.1230463981628418, "sampling/sampling_logp_difference/mean": 0.03234035521745682, "step": 664 }, { "clip_ratio/high_max": 0.0015432098880410194, "clip_ratio/high_mean": 0.0015432098880410194, "clip_ratio/low_mean": 0.009405238670296967, "clip_ratio/low_min": 0.009405238670296967, "clip_ratio/region_mean": 0.010948448558337986, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 80.625, "completions/mean_terminated_length": 80.625, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.07347342604771256, "epoch": 0.026710045386994417, "frac_reward_zero_std": 0.0, "grad_norm": 3.2300848960876465, "learning_rate": 7.987878787878789e-06, "loss": -0.0097, "num_tokens": 1443616.0, "reward": 0.7601395845413208, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9603564143180847, "reward_meter_std": 0.023673059418797493, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.16597026586532593, "reward_total_composite_mean": 0.7601395845413208, "reward_total_composite_std": 0.16597026586532593, "reward_total_mean": 0.7601395845413208, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9603564143180847, "rewards/meter/std": 0.023673059418797493, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7601395845413208, "rewards/total_composite/std": 0.16597026586532593, "sampling/importance_sampling_ratio/max": 1.564980387687683, "sampling/importance_sampling_ratio/mean": 1.0014386177062988, "sampling/importance_sampling_ratio/min": 0.3567086160182953, "sampling/sampling_logp_difference/max": 1.0308361053466797, "sampling/sampling_logp_difference/mean": 0.0124077582731843, "step": 665 }, { "clip_ratio/high_max": 0.006232782383449376, "clip_ratio/high_mean": 0.006232782383449376, "clip_ratio/low_mean": 0.0010416667209938169, "clip_ratio/low_min": 0.0010416667209938169, "clip_ratio/region_mean": 0.0072744491044431925, "completions/clipped_ratio": 0.0, "completions/max_length": 135.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 122.25, "completions/mean_terminated_length": 122.25, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.03877314692363143, "epoch": 0.02675021086877937, "frac_reward_zero_std": 0.0, "grad_norm": 3.160210609436035, "learning_rate": 7.984848484848486e-06, "loss": 0.0007, "num_tokens": 1445930.0, "reward": 0.12827160954475403, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8794082999229431, "reward_meter_std": 0.32782238721847534, "reward_repeat_penalty_mean": 0.212053582072258, "reward_repeat_penalty_std": 0.2030286192893982, "reward_std": 0.03277355059981346, "reward_total_composite_mean": 0.12827160954475403, "reward_total_composite_std": 0.03277355059981346, "reward_total_mean": 0.12827160954475403, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8794082999229431, "rewards/meter/std": 0.32782238721847534, "rewards/repeat_penalty/mean": 0.212053582072258, "rewards/repeat_penalty/std": 0.2030286192893982, "rewards/total_composite/mean": 0.12827160954475403, "rewards/total_composite/std": 0.03277355059981346, "sampling/importance_sampling_ratio/max": 1.6665557622909546, "sampling/importance_sampling_ratio/mean": 0.9980396032333374, "sampling/importance_sampling_ratio/min": 0.0016290009953081608, "sampling/sampling_logp_difference/max": 6.419788360595703, "sampling/sampling_logp_difference/mean": 0.02072002924978733, "step": 666 }, { "clip_ratio/high_max": 0.006372183095663786, "clip_ratio/high_mean": 0.006372183095663786, "clip_ratio/low_mean": 0.0031968391267582774, "clip_ratio/low_min": 0.0031968391267582774, "clip_ratio/region_mean": 0.009569022222422063, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 120.875, "completions/mean_terminated_length": 120.875, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.07529849279671907, "epoch": 0.026790376350564325, "frac_reward_zero_std": 0.0, "grad_norm": 4.775012016296387, "learning_rate": 7.981818181818183e-06, "loss": 0.0108, "num_tokens": 1448313.0, "reward": 0.24911293387413025, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9965612888336182, "reward_meter_std": 0.0018690497381612659, "reward_repeat_penalty_mean": 0.25, "reward_repeat_penalty_std": 0.21257823705673218, "reward_std": 0.2116239219903946, "reward_total_composite_mean": 0.24911293387413025, "reward_total_composite_std": 0.2116239219903946, "reward_total_mean": 0.24911293387413025, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9965612888336182, "rewards/meter/std": 0.0018690497381612659, "rewards/repeat_penalty/mean": 0.25, "rewards/repeat_penalty/std": 0.21257823705673218, "rewards/total_composite/mean": 0.24911293387413025, "rewards/total_composite/std": 0.2116239219903946, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9989076256752014, "sampling/importance_sampling_ratio/min": 0.3773258924484253, "sampling/sampling_logp_difference/max": 1.0500693321228027, "sampling/sampling_logp_difference/mean": 0.017011119052767754, "step": 667 }, { "clip_ratio/high_max": 0.0030251864809542894, "clip_ratio/high_mean": 0.0030251864809542894, "clip_ratio/low_mean": 0.007991038146428764, "clip_ratio/low_min": 0.007991038146428764, "clip_ratio/region_mean": 0.011016224627383053, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 207.0, "completions/mean_length": 243.0, "completions/mean_terminated_length": 204.57144165039062, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.05400602100417018, "epoch": 0.02683054183234928, "frac_reward_zero_std": 0.0, "grad_norm": 2.0396084785461426, "learning_rate": 7.978787878787879e-06, "loss": -0.1559, "num_tokens": 1451329.0, "reward": 0.17570531368255615, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9733182787895203, "reward_meter_std": 0.02887692302465439, "reward_repeat_penalty_mean": 0.27272728085517883, "reward_repeat_penalty_std": 0.13744163513183594, "reward_std": 0.11893083900213242, "reward_total_composite_mean": 0.17570531368255615, "reward_total_composite_std": 0.11893084645271301, "reward_total_mean": 0.17570531368255615, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9733182787895203, "rewards/meter/std": 0.02887692302465439, "rewards/repeat_penalty/mean": 0.27272728085517883, "rewards/repeat_penalty/std": 0.13744163513183594, "rewards/total_composite/mean": 0.17570531368255615, "rewards/total_composite/std": 0.11893084645271301, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9996070861816406, "sampling/importance_sampling_ratio/min": 0.1251610666513443, "sampling/sampling_logp_difference/max": 2.0781538486480713, "sampling/sampling_logp_difference/mean": 0.01857808604836464, "step": 668 }, { "clip_ratio/high_max": 0.011646514758467674, "clip_ratio/high_mean": 0.011646514758467674, "clip_ratio/low_mean": 0.009377967799082398, "clip_ratio/low_min": 0.009377967799082398, "clip_ratio/region_mean": 0.021024482557550073, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 240.125, "completions/mean_terminated_length": 77.0, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.16993126086890697, "epoch": 0.026870707314134233, "frac_reward_zero_std": 0.0, "grad_norm": 2.1599395275115967, "learning_rate": 7.975757575757576e-06, "loss": -0.023, "num_tokens": 1453130.0, "reward": 0.4261804223060608, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5011062026023865, "reward_meter_std": 0.4051038920879364, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.46671321988105774, "reward_total_composite_mean": 0.4261804223060608, "reward_total_composite_std": 0.4667132496833801, "reward_total_mean": 0.4261804223060608, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5011062026023865, "rewards/meter/std": 0.4051038920879364, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4261804223060608, "rewards/total_composite/std": 0.4667132496833801, "sampling/importance_sampling_ratio/max": 1.8760875463485718, "sampling/importance_sampling_ratio/mean": 1.0080572366714478, "sampling/importance_sampling_ratio/min": 0.16133907437324524, "sampling/sampling_logp_difference/max": 1.824247121810913, "sampling/sampling_logp_difference/mean": 0.052589599043130875, "step": 669 }, { "clip_ratio/high_max": 0.013186233583837748, "clip_ratio/high_mean": 0.013186233583837748, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.013186233583837748, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 351.0, "completions/mean_terminated_length": 82.66667175292969, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.09451978467404842, "epoch": 0.026910872795919187, "frac_reward_zero_std": 0.0, "grad_norm": 0.766589879989624, "learning_rate": 7.972727272727273e-06, "loss": -0.1072, "num_tokens": 1454698.0, "reward": 0.249087393283844, "reward_arabic_clean_mean": 0.375, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6174333691596985, "reward_meter_std": 0.4055008888244629, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_std": 0.3437734842300415, "reward_total_composite_mean": 0.249087393283844, "reward_total_composite_std": 0.3437734544277191, "reward_total_mean": 0.249087393283844, "rewards/arabic_clean/mean": 0.375, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6174333691596985, "rewards/meter/std": 0.4055008888244629, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.249087393283844, "rewards/total_composite/std": 0.3437734544277191, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004920482635498, "sampling/importance_sampling_ratio/min": 0.4397118091583252, "sampling/sampling_logp_difference/max": 1.1255141496658325, "sampling/sampling_logp_difference/mean": 0.03491988405585289, "step": 670 }, { "clip_ratio/high_max": 0.006735671544447541, "clip_ratio/high_mean": 0.006735671544447541, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006735671544447541, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 354.75, "completions/mean_terminated_length": 92.66667175292969, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.07205967605113983, "epoch": 0.02695103827770414, "frac_reward_zero_std": 0.0, "grad_norm": 0.7819209098815918, "learning_rate": 7.96969696969697e-06, "loss": -0.0912, "num_tokens": 1456328.0, "reward": 0.19879300892353058, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.7872094511985779, "reward_meter_std": 0.3365079164505005, "reward_repeat_penalty_mean": 0.4750000238418579, "reward_repeat_penalty_std": 0.2121320366859436, "reward_std": 0.21251867711544037, "reward_total_composite_mean": 0.19879300892353058, "reward_total_composite_std": 0.21251867711544037, "reward_total_mean": 0.19879300892353058, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.7872094511985779, "rewards/meter/std": 0.3365079164505005, "rewards/repeat_penalty/mean": 0.4750000238418579, "rewards/repeat_penalty/std": 0.2121320366859436, "rewards/total_composite/mean": 0.19879300892353058, "rewards/total_composite/std": 0.21251867711544037, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0077892541885376, "sampling/importance_sampling_ratio/min": 0.31241610646247864, "sampling/sampling_logp_difference/max": 1.163419246673584, "sampling/sampling_logp_difference/mean": 0.0315895713865757, "step": 671 }, { "clip_ratio/high_max": 0.010866477387025952, "clip_ratio/high_mean": 0.010866477387025952, "clip_ratio/low_mean": 0.0028696630615741014, "clip_ratio/low_min": 0.0028696630615741014, "clip_ratio/region_mean": 0.013736140448600054, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 220.0, "completions/mean_length": 314.0, "completions/mean_terminated_length": 195.1999969482422, "completions/min_length": 160.0, "completions/min_terminated_length": 160.0, "entropy": 0.06148350611329079, "epoch": 0.026991203759489095, "frac_reward_zero_std": 0.0, "grad_norm": 0.8036666512489319, "learning_rate": 7.966666666666668e-06, "loss": -0.1237, "num_tokens": 1458728.0, "reward": 0.23602712154388428, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.1414213627576828, "reward_meter_mean": 0.7117201685905457, "reward_meter_std": 0.3770325481891632, "reward_repeat_penalty_mean": 0.5681818127632141, "reward_repeat_penalty_std": 0.2368127554655075, "reward_std": 0.1749935895204544, "reward_total_composite_mean": 0.23602712154388428, "reward_total_composite_std": 0.1749935895204544, "reward_total_mean": 0.23602712154388428, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.1414213627576828, "rewards/meter/mean": 0.7117201685905457, "rewards/meter/std": 0.3770325481891632, "rewards/repeat_penalty/mean": 0.5681818127632141, "rewards/repeat_penalty/std": 0.2368127554655075, "rewards/total_composite/mean": 0.23602712154388428, "rewards/total_composite/std": 0.1749935895204544, "sampling/importance_sampling_ratio/max": 1.857919692993164, "sampling/importance_sampling_ratio/mean": 0.9994598627090454, "sampling/importance_sampling_ratio/min": 0.22612299025058746, "sampling/sampling_logp_difference/max": 1.4866762161254883, "sampling/sampling_logp_difference/mean": 0.020951304584741592, "step": 672 }, { "clip_ratio/high_max": 0.02129602595232427, "clip_ratio/high_mean": 0.02129602595232427, "clip_ratio/low_mean": 0.020833334419876337, "clip_ratio/low_min": 0.020833334419876337, "clip_ratio/region_mean": 0.04212936037220061, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 117.625, "completions/mean_terminated_length": 61.28571701049805, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.2579981219023466, "epoch": 0.02703136924127405, "frac_reward_zero_std": 0.0, "grad_norm": 4.668978214263916, "learning_rate": 7.963636363636365e-06, "loss": -0.0271, "num_tokens": 1460405.0, "reward": 0.768899142742157, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.8629885911941528, "reward_meter_std": 0.28332778811454773, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.32128065824508667, "reward_total_composite_mean": 0.768899142742157, "reward_total_composite_std": 0.3212806284427643, "reward_total_mean": 0.768899142742157, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.8629885911941528, "rewards/meter/std": 0.28332778811454773, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.768899142742157, "rewards/total_composite/std": 0.3212806284427643, "sampling/importance_sampling_ratio/max": 1.7451122999191284, "sampling/importance_sampling_ratio/mean": 1.0077714920043945, "sampling/importance_sampling_ratio/min": 0.1919175386428833, "sampling/sampling_logp_difference/max": 1.6506894826889038, "sampling/sampling_logp_difference/mean": 0.04510435834527016, "step": 673 }, { "clip_ratio/high_max": 0.04135313397273421, "clip_ratio/high_mean": 0.04135313397273421, "clip_ratio/low_mean": 0.009934040834195912, "clip_ratio/low_min": 0.009934040834195912, "clip_ratio/region_mean": 0.051287174806930125, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 128.875, "completions/mean_terminated_length": 74.14286041259766, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.23062355443835258, "epoch": 0.027071534723059003, "frac_reward_zero_std": 0.0, "grad_norm": 3.691714286804199, "learning_rate": 7.96060606060606e-06, "loss": -0.0956, "num_tokens": 1462244.0, "reward": 0.5691444873809814, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7070336937904358, "reward_meter_std": 0.39935943484306335, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_std": 0.3437305986881256, "reward_total_composite_mean": 0.5691444873809814, "reward_total_composite_std": 0.343730628490448, "reward_total_mean": 0.5691444873809814, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7070336937904358, "rewards/meter/std": 0.39935943484306335, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.5691444873809814, "rewards/total_composite/std": 0.343730628490448, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0029557943344116, "sampling/importance_sampling_ratio/min": 0.28025004267692566, "sampling/sampling_logp_difference/max": 1.2720730304718018, "sampling/sampling_logp_difference/mean": 0.05939958617091179, "step": 674 }, { "clip_ratio/high_max": 0.000681198900565505, "clip_ratio/high_mean": 0.000681198900565505, "clip_ratio/low_mean": 0.0032165506563615054, "clip_ratio/low_min": 0.0032165506563615054, "clip_ratio/region_mean": 0.0038977495569270104, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 376.125, "completions/mean_terminated_length": 356.71429443359375, "completions/min_length": 337.0, "completions/min_terminated_length": 337.0, "entropy": 0.0205289286095649, "epoch": 0.027111700204843957, "frac_reward_zero_std": 0.0, "grad_norm": 0.7685384154319763, "learning_rate": 7.957575757575758e-06, "loss": -0.0973, "num_tokens": 1466421.0, "reward": 0.15836204588413239, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8624999523162842, "reward_count_adherence_std": 0.31139087677001953, "reward_meter_mean": 0.9768602848052979, "reward_meter_std": 0.0202656090259552, "reward_repeat_penalty_mean": 0.2991071343421936, "reward_repeat_penalty_std": 0.36836329102516174, "reward_std": 0.21933621168136597, "reward_total_composite_mean": 0.15836204588413239, "reward_total_composite_std": 0.21933622658252716, "reward_total_mean": 0.15836204588413239, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8624999523162842, "rewards/count_adherence/std": 0.31139087677001953, "rewards/meter/mean": 0.9768602848052979, "rewards/meter/std": 0.0202656090259552, "rewards/repeat_penalty/mean": 0.2991071343421936, "rewards/repeat_penalty/std": 0.36836329102516174, "rewards/total_composite/mean": 0.15836204588413239, "rewards/total_composite/std": 0.21933622658252716, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0009857416152954, "sampling/importance_sampling_ratio/min": 0.4014633893966675, "sampling/sampling_logp_difference/max": 1.467843770980835, "sampling/sampling_logp_difference/mean": 0.005116640590131283, "step": 675 }, { "clip_ratio/high_max": 0.01704174862243235, "clip_ratio/high_mean": 0.01704174862243235, "clip_ratio/low_mean": 0.009073751280084252, "clip_ratio/low_min": 0.009073751280084252, "clip_ratio/region_mean": 0.026115499902516603, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 45.5, "completions/mean_terminated_length": 45.5, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.16463111247867346, "epoch": 0.02715186568662891, "frac_reward_zero_std": 0.0, "grad_norm": 9.489598274230957, "learning_rate": 7.954545454545455e-06, "loss": -0.1218, "num_tokens": 1468025.0, "reward": 0.9916675090789795, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9916675090789795, "reward_meter_std": 0.0027789692394435406, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002778968308120966, "reward_total_composite_mean": 0.9916675090789795, "reward_total_composite_std": 0.0027789692394435406, "reward_total_mean": 0.9916675090789795, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9916675090789795, "rewards/meter/std": 0.0027789692394435406, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9916675090789795, "rewards/total_composite/std": 0.0027789692394435406, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9964287877082825, "sampling/importance_sampling_ratio/min": 0.13002057373523712, "sampling/sampling_logp_difference/max": 2.040062665939331, "sampling/sampling_logp_difference/mean": 0.040273357182741165, "step": 676 }, { "clip_ratio/high_max": 0.0021551724057644606, "clip_ratio/high_mean": 0.0021551724057644606, "clip_ratio/low_mean": 0.012098872568458319, "clip_ratio/low_min": 0.012098872568458319, "clip_ratio/region_mean": 0.01425404497422278, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 62.125, "completions/mean_terminated_length": 62.125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.11069364938884974, "epoch": 0.027192031168413865, "frac_reward_zero_std": 0.0, "grad_norm": 4.125744342803955, "learning_rate": 7.951515151515152e-06, "loss": 0.0185, "num_tokens": 1469690.0, "reward": 0.6625509858131409, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9938265085220337, "reward_meter_std": 0.0003937912406399846, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002625406195875257, "reward_total_composite_mean": 0.6625509858131409, "reward_total_composite_std": 0.00026252749375998974, "reward_total_mean": 0.6625509858131409, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9938265085220337, "rewards/meter/std": 0.0003937912406399846, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6625509858131409, "rewards/total_composite/std": 0.00026252749375998974, "sampling/importance_sampling_ratio/max": 1.5561105012893677, "sampling/importance_sampling_ratio/mean": 1.0004545450210571, "sampling/importance_sampling_ratio/min": 0.0030335744377225637, "sampling/sampling_logp_difference/max": 5.798013687133789, "sampling/sampling_logp_difference/mean": 0.02882186695933342, "step": 677 }, { "clip_ratio/high_max": 0.002251501166028902, "clip_ratio/high_mean": 0.002251501166028902, "clip_ratio/low_mean": 0.004044867469929159, "clip_ratio/low_min": 0.004044867469929159, "clip_ratio/region_mean": 0.006296368635958061, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 291.0, "completions/mean_length": 336.25, "completions/mean_terminated_length": 277.66668701171875, "completions/min_length": 268.0, "completions/min_terminated_length": 268.0, "entropy": 0.03139376197941601, "epoch": 0.02723219665019882, "frac_reward_zero_std": 0.0, "grad_norm": 0.7712501287460327, "learning_rate": 7.948484848484848e-06, "loss": -0.15, "num_tokens": 1472788.0, "reward": 0.1397983729839325, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8035714626312256, "reward_count_adherence_std": 0.15152288973331451, "reward_meter_mean": 0.6280694007873535, "reward_meter_std": 0.5047115683555603, "reward_repeat_penalty_mean": 0.41633522510528564, "reward_repeat_penalty_std": 0.3482770323753357, "reward_std": 0.1796947568655014, "reward_total_composite_mean": 0.1397983729839325, "reward_total_composite_std": 0.1796947717666626, "reward_total_mean": 0.1397983729839325, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8035714626312256, "rewards/count_adherence/std": 0.15152288973331451, "rewards/meter/mean": 0.6280694007873535, "rewards/meter/std": 0.5047115683555603, "rewards/repeat_penalty/mean": 0.41633522510528564, "rewards/repeat_penalty/std": 0.3482770323753357, "rewards/total_composite/mean": 0.1397983729839325, "rewards/total_composite/std": 0.1796947717666626, "sampling/importance_sampling_ratio/max": 1.6066981554031372, "sampling/importance_sampling_ratio/mean": 0.9978663921356201, "sampling/importance_sampling_ratio/min": 0.003241149475798011, "sampling/sampling_logp_difference/max": 5.731827259063721, "sampling/sampling_logp_difference/mean": 0.01377029623836279, "step": 678 }, { "clip_ratio/high_max": 0.01512786210514605, "clip_ratio/high_mean": 0.01512786210514605, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/region_mean": 0.017486352706328034, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 57.0, "completions/mean_terminated_length": 57.0, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.11628487333655357, "epoch": 0.027272362131983773, "frac_reward_zero_std": 0.0, "grad_norm": 12.517946243286133, "learning_rate": 7.945454545454547e-06, "loss": -0.0123, "num_tokens": 1474604.0, "reward": 0.9428657293319702, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9841916561126709, "reward_meter_std": 0.005828971043229103, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.1139112189412117, "reward_total_composite_mean": 0.9428657293319702, "reward_total_composite_std": 0.1139112040400505, "reward_total_mean": 0.9428657293319702, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9841916561126709, "rewards/meter/std": 0.005828971043229103, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9428657293319702, "rewards/total_composite/std": 0.1139112040400505, "sampling/importance_sampling_ratio/max": 1.880238652229309, "sampling/importance_sampling_ratio/mean": 0.9998968243598938, "sampling/importance_sampling_ratio/min": 0.3291391432285309, "sampling/sampling_logp_difference/max": 1.1112747192382812, "sampling/sampling_logp_difference/mean": 0.02025281824171543, "step": 679 }, { "clip_ratio/high_max": 0.0033385155256837606, "clip_ratio/high_mean": 0.0033385155256837606, "clip_ratio/low_mean": 0.014545571291819215, "clip_ratio/low_min": 0.014545571291819215, "clip_ratio/region_mean": 0.017884086817502975, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 115.625, "completions/mean_terminated_length": 115.625, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.08438334474340081, "epoch": 0.027312527613768726, "frac_reward_zero_std": 0.0, "grad_norm": 8.33311653137207, "learning_rate": 7.942424242424242e-06, "loss": 0.031, "num_tokens": 1476777.0, "reward": 0.22119268774986267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.371232271194458, "reward_meter_std": 0.47603076696395874, "reward_repeat_penalty_mean": 0.48125001788139343, "reward_repeat_penalty_std": 0.13611315190792084, "reward_std": 0.2868722081184387, "reward_total_composite_mean": 0.22119268774986267, "reward_total_composite_std": 0.28687217831611633, "reward_total_mean": 0.22119268774986267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.371232271194458, "rewards/meter/std": 0.47603076696395874, "rewards/repeat_penalty/mean": 0.48125001788139343, "rewards/repeat_penalty/std": 0.13611315190792084, "rewards/total_composite/mean": 0.22119268774986267, "rewards/total_composite/std": 0.28687217831611633, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0001816749572754, "sampling/importance_sampling_ratio/min": 0.1354907900094986, "sampling/sampling_logp_difference/max": 1.9988516569137573, "sampling/sampling_logp_difference/mean": 0.020960384979844093, "step": 680 }, { "clip_ratio/high_max": 0.015127861872315407, "clip_ratio/high_mean": 0.015127861872315407, "clip_ratio/low_mean": 0.014689265750348568, "clip_ratio/low_min": 0.014689265750348568, "clip_ratio/region_mean": 0.029817127622663975, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 57.875, "completions/mean_terminated_length": 57.875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.11835484858602285, "epoch": 0.02735269309555368, "frac_reward_zero_std": 0.0, "grad_norm": 3.9222095012664795, "learning_rate": 7.93939393939394e-06, "loss": 0.0231, "num_tokens": 1478560.0, "reward": 0.8934845924377441, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9723160266876221, "reward_meter_std": 0.027899622917175293, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.163389652967453, "reward_total_composite_mean": 0.8934845924377441, "reward_total_composite_std": 0.1633896678686142, "reward_total_mean": 0.8934845924377441, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9723160266876221, "rewards/meter/std": 0.027899622917175293, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8934845924377441, "rewards/total_composite/std": 0.1633896678686142, "sampling/importance_sampling_ratio/max": 1.6929512023925781, "sampling/importance_sampling_ratio/mean": 1.0034892559051514, "sampling/importance_sampling_ratio/min": 0.342024028301239, "sampling/sampling_logp_difference/max": 1.0728743076324463, "sampling/sampling_logp_difference/mean": 0.023344971239566803, "step": 681 }, { "clip_ratio/high_max": 0.0029069767333567142, "clip_ratio/high_mean": 0.0029069767333567142, "clip_ratio/low_mean": 0.005251961061730981, "clip_ratio/low_min": 0.005251961061730981, "clip_ratio/region_mean": 0.008158937795087695, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 173.875, "completions/mean_terminated_length": 125.5714340209961, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.049124513287097216, "epoch": 0.027392858577338634, "frac_reward_zero_std": 0.0, "grad_norm": 1.2039419412612915, "learning_rate": 7.936363636363637e-06, "loss": -0.0676, "num_tokens": 1480663.0, "reward": 0.16035160422325134, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.2519763112068176, "reward_meter_mean": 0.38672682642936707, "reward_meter_std": 0.36102262139320374, "reward_repeat_penalty_mean": 0.48750001192092896, "reward_repeat_penalty_std": 0.24604006111621857, "reward_std": 0.24009782075881958, "reward_total_composite_mean": 0.16035160422325134, "reward_total_composite_std": 0.24009783565998077, "reward_total_mean": 0.16035160422325134, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.2519763112068176, "rewards/meter/mean": 0.38672682642936707, "rewards/meter/std": 0.36102262139320374, "rewards/repeat_penalty/mean": 0.48750001192092896, "rewards/repeat_penalty/std": 0.24604006111621857, "rewards/total_composite/mean": 0.16035160422325134, "rewards/total_composite/std": 0.24009783565998077, "sampling/importance_sampling_ratio/max": 1.6329307556152344, "sampling/importance_sampling_ratio/mean": 1.0000630617141724, "sampling/importance_sampling_ratio/min": 0.29752659797668457, "sampling/sampling_logp_difference/max": 1.2122516632080078, "sampling/sampling_logp_difference/mean": 0.013726767152547836, "step": 682 }, { "clip_ratio/high_max": 0.016894312808290124, "clip_ratio/high_mean": 0.016894312808290124, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/region_mean": 0.02113160095177591, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 58.75, "completions/mean_terminated_length": 58.75, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.11382754053920507, "epoch": 0.02743302405912359, "frac_reward_zero_std": 0.0, "grad_norm": 4.130276679992676, "learning_rate": 7.933333333333334e-06, "loss": 0.0096, "num_tokens": 1482349.0, "reward": 0.8474001884460449, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9239417314529419, "reward_meter_std": 0.05428864806890488, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.15459680557250977, "reward_total_composite_mean": 0.8474001884460449, "reward_total_composite_std": 0.15459680557250977, "reward_total_mean": 0.8474001884460449, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9239417314529419, "rewards/meter/std": 0.05428864806890488, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8474001884460449, "rewards/total_composite/std": 0.15459680557250977, "sampling/importance_sampling_ratio/max": 1.6103789806365967, "sampling/importance_sampling_ratio/mean": 0.994130551815033, "sampling/importance_sampling_ratio/min": 0.15196943283081055, "sampling/sampling_logp_difference/max": 1.8840758800506592, "sampling/sampling_logp_difference/mean": 0.03419341892004013, "step": 683 }, { "clip_ratio/high_max": 0.004629629664123058, "clip_ratio/high_mean": 0.004629629664123058, "clip_ratio/low_mean": 0.007758883642964065, "clip_ratio/low_min": 0.007758883642964065, "clip_ratio/region_mean": 0.012388513307087123, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 80.25, "completions/mean_terminated_length": 80.25, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.0625843945890665, "epoch": 0.027473189540908542, "frac_reward_zero_std": 0.0, "grad_norm": 4.445869445800781, "learning_rate": 7.930303030303031e-06, "loss": -0.0064, "num_tokens": 1484327.0, "reward": 0.46393126249313354, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9331251978874207, "reward_meter_std": 0.07388782501220703, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.10690450668334961, "reward_std": 0.09499124437570572, "reward_total_composite_mean": 0.46393126249313354, "reward_total_composite_std": 0.09499123692512512, "reward_total_mean": 0.46393126249313354, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9331251978874207, "rewards/meter/std": 0.07388782501220703, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.10690450668334961, "rewards/total_composite/mean": 0.46393126249313354, "rewards/total_composite/std": 0.09499123692512512, "sampling/importance_sampling_ratio/max": 1.7777718305587769, "sampling/importance_sampling_ratio/mean": 1.0021919012069702, "sampling/importance_sampling_ratio/min": 0.3052648901939392, "sampling/sampling_logp_difference/max": 1.1865754127502441, "sampling/sampling_logp_difference/mean": 0.01861550658941269, "step": 684 }, { "clip_ratio/high_max": 0.010335821425542235, "clip_ratio/high_mean": 0.010335821425542235, "clip_ratio/low_mean": 0.0007022471982054412, "clip_ratio/low_min": 0.0007022471982054412, "clip_ratio/region_mean": 0.011038068623747677, "completions/clipped_ratio": 0.0, "completions/max_length": 178.0, "completions/max_terminated_length": 178.0, "completions/mean_length": 170.625, "completions/mean_terminated_length": 170.625, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.046010758727788925, "epoch": 0.027513355022693496, "frac_reward_zero_std": 0.0, "grad_norm": 2.300926685333252, "learning_rate": 7.927272727272729e-06, "loss": 0.0202, "num_tokens": 1487252.0, "reward": 0.34419798851013184, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8037871718406677, "reward_meter_std": 0.32751503586769104, "reward_repeat_penalty_mean": 0.4107142686843872, "reward_repeat_penalty_std": 0.050507623702287674, "reward_std": 0.14113974571228027, "reward_total_composite_mean": 0.34419798851013184, "reward_total_composite_std": 0.14113976061344147, "reward_total_mean": 0.34419798851013184, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8037871718406677, "rewards/meter/std": 0.32751503586769104, "rewards/repeat_penalty/mean": 0.4107142686843872, "rewards/repeat_penalty/std": 0.050507623702287674, "rewards/total_composite/mean": 0.34419798851013184, "rewards/total_composite/std": 0.14113976061344147, "sampling/importance_sampling_ratio/max": 1.4589576721191406, "sampling/importance_sampling_ratio/mean": 0.9983497262001038, "sampling/importance_sampling_ratio/min": 0.19332465529441833, "sampling/sampling_logp_difference/max": 1.643384337425232, "sampling/sampling_logp_difference/mean": 0.010442078113555908, "step": 685 }, { "clip_ratio/high_max": 0.00234195904340595, "clip_ratio/high_mean": 0.00234195904340595, "clip_ratio/low_mean": 0.003458057180978358, "clip_ratio/low_min": 0.003458057180978358, "clip_ratio/region_mean": 0.005800016224384308, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 222.0, "completions/mean_length": 251.75, "completions/mean_terminated_length": 214.57144165039062, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.021744283847510815, "epoch": 0.02755352050447845, "frac_reward_zero_std": 0.0, "grad_norm": 1.047049641609192, "learning_rate": 7.924242424242426e-06, "loss": -0.0998, "num_tokens": 1490154.0, "reward": 0.2209167182445526, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.2828427255153656, "reward_meter_mean": 0.9798320531845093, "reward_meter_std": 0.047762468457221985, "reward_repeat_penalty_mean": 0.32499998807907104, "reward_repeat_penalty_std": 0.27645719051361084, "reward_std": 0.04932519420981407, "reward_total_composite_mean": 0.2209167182445526, "reward_total_composite_std": 0.04932519793510437, "reward_total_mean": 0.2209167182445526, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.2828427255153656, "rewards/meter/mean": 0.9798320531845093, "rewards/meter/std": 0.047762468457221985, "rewards/repeat_penalty/mean": 0.32499998807907104, "rewards/repeat_penalty/std": 0.27645719051361084, "rewards/total_composite/mean": 0.2209167182445526, "rewards/total_composite/std": 0.04932519793510437, "sampling/importance_sampling_ratio/max": 1.7827941179275513, "sampling/importance_sampling_ratio/mean": 0.9986944198608398, "sampling/importance_sampling_ratio/min": 0.053416479378938675, "sampling/sampling_logp_difference/max": 2.929636001586914, "sampling/sampling_logp_difference/mean": 0.011242986656725407, "step": 686 }, { "clip_ratio/high_max": 0.020970338257029653, "clip_ratio/high_mean": 0.020970338257029653, "clip_ratio/low_mean": 0.011955027701333165, "clip_ratio/low_min": 0.011955027701333165, "clip_ratio/region_mean": 0.03292536595836282, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 71.0, "completions/mean_terminated_length": 71.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.1976525131613016, "epoch": 0.027593685986263404, "frac_reward_zero_std": 0.0, "grad_norm": 10.655113220214844, "learning_rate": 7.921212121212122e-06, "loss": 0.0292, "num_tokens": 1491874.0, "reward": 0.8430095314979553, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8845878839492798, "reward_meter_std": 0.2909509837627411, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.2961682677268982, "reward_total_composite_mean": 0.8430095314979553, "reward_total_composite_std": 0.2961682975292206, "reward_total_mean": 0.8430095314979553, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8845878839492798, "rewards/meter/std": 0.2909509837627411, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8430095314979553, "rewards/total_composite/std": 0.2961682975292206, "sampling/importance_sampling_ratio/max": 1.9307304620742798, "sampling/importance_sampling_ratio/mean": 1.0036027431488037, "sampling/importance_sampling_ratio/min": 0.05301474407315254, "sampling/sampling_logp_difference/max": 2.937185287475586, "sampling/sampling_logp_difference/mean": 0.05931292846798897, "step": 687 }, { "clip_ratio/high_max": 0.010716472752392292, "clip_ratio/high_mean": 0.010716472752392292, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/region_mean": 0.02208010945469141, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 67.75, "completions/mean_terminated_length": 67.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.06369170360267162, "epoch": 0.027633851468048358, "frac_reward_zero_std": 0.0, "grad_norm": 1.4717203378677368, "learning_rate": 7.918181818181819e-06, "loss": -0.0086, "num_tokens": 1493736.0, "reward": 0.6656583547592163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984875321388245, "reward_meter_std": 0.0001689638738753274, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011264239583397284, "reward_total_composite_mean": 0.6656583547592163, "reward_total_composite_std": 0.00011263116175541654, "reward_total_mean": 0.6656583547592163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984875321388245, "rewards/meter/std": 0.0001689638738753274, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6656583547592163, "rewards/total_composite/std": 0.00011263116175541654, "sampling/importance_sampling_ratio/max": 1.6862038373947144, "sampling/importance_sampling_ratio/mean": 0.9985640048980713, "sampling/importance_sampling_ratio/min": 0.07396066933870316, "sampling/sampling_logp_difference/max": 2.604221820831299, "sampling/sampling_logp_difference/mean": 0.018908310681581497, "step": 688 }, { "clip_ratio/high_max": 0.0019430051324889064, "clip_ratio/high_mean": 0.0019430051324889064, "clip_ratio/low_mean": 0.007414351915940642, "clip_ratio/low_min": 0.007414351915940642, "clip_ratio/region_mean": 0.009357357048429549, "completions/clipped_ratio": 0.0, "completions/max_length": 212.0, "completions/max_terminated_length": 212.0, "completions/mean_length": 202.625, "completions/mean_terminated_length": 202.625, "completions/min_length": 191.0, "completions/min_terminated_length": 191.0, "entropy": 0.03671248443424702, "epoch": 0.027674016949833312, "frac_reward_zero_std": 0.0, "grad_norm": 2.0693576335906982, "learning_rate": 7.915151515151516e-06, "loss": 0.0309, "num_tokens": 1497021.0, "reward": 0.4050329029560089, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.995306134223938, "reward_meter_std": 0.0053658634424209595, "reward_repeat_penalty_mean": 0.4194444417953491, "reward_repeat_penalty_std": 0.13975918292999268, "reward_std": 0.1355670541524887, "reward_total_composite_mean": 0.4050329029560089, "reward_total_composite_std": 0.1355670541524887, "reward_total_mean": 0.4050329029560089, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.995306134223938, "rewards/meter/std": 0.0053658634424209595, "rewards/repeat_penalty/mean": 0.4194444417953491, "rewards/repeat_penalty/std": 0.13975918292999268, "rewards/total_composite/mean": 0.4050329029560089, "rewards/total_composite/std": 0.1355670541524887, "sampling/importance_sampling_ratio/max": 1.674434781074524, "sampling/importance_sampling_ratio/mean": 0.9986175298690796, "sampling/importance_sampling_ratio/min": 0.08943047374486923, "sampling/sampling_logp_difference/max": 2.4142937660217285, "sampling/sampling_logp_difference/mean": 0.014271133579313755, "step": 689 }, { "clip_ratio/high_max": 0.0015578281017951667, "clip_ratio/high_mean": 0.0015578281017951667, "clip_ratio/low_mean": 0.0021797263179905713, "clip_ratio/low_min": 0.0021797263179905713, "clip_ratio/region_mean": 0.003737554419785738, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 401.625, "completions/mean_terminated_length": 401.625, "completions/min_length": 400.0, "completions/min_terminated_length": 400.0, "entropy": 0.027160495053976774, "epoch": 0.027714182431618266, "frac_reward_zero_std": 0.0, "grad_norm": 1.1219055652618408, "learning_rate": 7.912121212121213e-06, "loss": 0.0024, "num_tokens": 1502074.0, "reward": 0.1784874051809311, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981971979141235, "reward_meter_std": 0.000627980858553201, "reward_repeat_penalty_mean": 0.2503289580345154, "reward_repeat_penalty_std": 0.23873279988765717, "reward_std": 0.17026525735855103, "reward_total_composite_mean": 0.1784874051809311, "reward_total_composite_std": 0.17026527225971222, "reward_total_mean": 0.1784874051809311, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981971979141235, "rewards/meter/std": 0.000627980858553201, "rewards/repeat_penalty/mean": 0.2503289580345154, "rewards/repeat_penalty/std": 0.23873279988765717, "rewards/total_composite/mean": 0.1784874051809311, "rewards/total_composite/std": 0.17026527225971222, "sampling/importance_sampling_ratio/max": 1.6704438924789429, "sampling/importance_sampling_ratio/mean": 1.0005240440368652, "sampling/importance_sampling_ratio/min": 0.24811138212680817, "sampling/sampling_logp_difference/max": 1.3938775062561035, "sampling/sampling_logp_difference/mean": 0.005141077097505331, "step": 690 }, { "clip_ratio/high_max": 0.04530784301459789, "clip_ratio/high_mean": 0.04530784301459789, "clip_ratio/low_mean": 0.019354344811290503, "clip_ratio/low_min": 0.019354344811290503, "clip_ratio/region_mean": 0.0646621878258884, "completions/clipped_ratio": 0.0, "completions/max_length": 44.0, "completions/max_terminated_length": 44.0, "completions/mean_length": 38.125, "completions/mean_terminated_length": 38.125, "completions/min_length": 19.0, "completions/min_terminated_length": 19.0, "entropy": 0.5944820679724216, "epoch": 0.02775434791340322, "frac_reward_zero_std": 0.0, "grad_norm": 11.69098949432373, "learning_rate": 7.909090909090909e-06, "loss": -0.1031, "num_tokens": 1503707.0, "reward": 0.4370739161968231, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.4370739161968231, "reward_meter_std": 0.2700711488723755, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2700711488723755, "reward_total_composite_mean": 0.4370739161968231, "reward_total_composite_std": 0.2700711488723755, "reward_total_mean": 0.4370739161968231, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.4370739161968231, "rewards/meter/std": 0.2700711488723755, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4370739161968231, "rewards/total_composite/std": 0.2700711488723755, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0089696645736694, "sampling/importance_sampling_ratio/min": 0.3319256007671356, "sampling/sampling_logp_difference/max": 1.102844476699829, "sampling/sampling_logp_difference/mean": 0.0835486352443695, "step": 691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0014367816038429737, "clip_ratio/low_min": 0.0014367816038429737, "clip_ratio/region_mean": 0.0014367816038429737, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 86.625, "completions/mean_terminated_length": 86.625, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.04205932654440403, "epoch": 0.027794513395188174, "frac_reward_zero_std": 0.0, "grad_norm": 5.031876564025879, "learning_rate": 7.906060606060608e-06, "loss": 0.0153, "num_tokens": 1505712.0, "reward": 0.9942313432693481, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942313432693481, "reward_meter_std": 0.0012215422466397285, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00122154806740582, "reward_total_composite_mean": 0.9942313432693481, "reward_total_composite_std": 0.0012215422466397285, "reward_total_mean": 0.9942313432693481, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942313432693481, "rewards/meter/std": 0.0012215422466397285, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942313432693481, "rewards/total_composite/std": 0.0012215422466397285, "sampling/importance_sampling_ratio/max": 1.5647001266479492, "sampling/importance_sampling_ratio/mean": 1.003043293952942, "sampling/importance_sampling_ratio/min": 0.7906866669654846, "sampling/sampling_logp_difference/max": 0.44769424200057983, "sampling/sampling_logp_difference/mean": 0.005889675114303827, "step": 692 }, { "clip_ratio/high_max": 0.01416083937510848, "clip_ratio/high_mean": 0.01416083937510848, "clip_ratio/low_mean": 0.009437322150915861, "clip_ratio/low_min": 0.009437322150915861, "clip_ratio/region_mean": 0.02359816152602434, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 54.125, "completions/mean_terminated_length": 54.125, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.0895584006793797, "epoch": 0.027834678876973128, "frac_reward_zero_std": 0.0, "grad_norm": 10.364151954650879, "learning_rate": 7.903030303030303e-06, "loss": 0.0041, "num_tokens": 1507385.0, "reward": 0.9447178244590759, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9447178244590759, "reward_meter_std": 0.019572317600250244, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01957232505083084, "reward_total_composite_mean": 0.9447178244590759, "reward_total_composite_std": 0.019572317600250244, "reward_total_mean": 0.9447178244590759, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9447178244590759, "rewards/meter/std": 0.019572317600250244, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9447178244590759, "rewards/total_composite/std": 0.019572317600250244, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003846287727356, "sampling/importance_sampling_ratio/min": 0.14936968684196472, "sampling/sampling_logp_difference/max": 1.9013309478759766, "sampling/sampling_logp_difference/mean": 0.024356268346309662, "step": 693 }, { "clip_ratio/high_max": 0.02390445303171873, "clip_ratio/high_mean": 0.02390445303171873, "clip_ratio/low_mean": 0.01502489356789738, "clip_ratio/low_min": 0.01502489356789738, "clip_ratio/region_mean": 0.03892934659961611, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 74.0, "completions/mean_terminated_length": 74.0, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.17548873648047447, "epoch": 0.02787484435875808, "frac_reward_zero_std": 0.0, "grad_norm": 6.670380115509033, "learning_rate": 7.9e-06, "loss": 0.0035, "num_tokens": 1509273.0, "reward": 0.5444035530090332, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.610145092010498, "reward_meter_std": 0.23973627388477325, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.24921227991580963, "reward_total_composite_mean": 0.5444035530090332, "reward_total_composite_std": 0.24921227991580963, "reward_total_mean": 0.5444035530090332, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.610145092010498, "rewards/meter/std": 0.23973627388477325, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.5444035530090332, "rewards/total_composite/std": 0.24921227991580963, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0047415494918823, "sampling/importance_sampling_ratio/min": 0.16379445791244507, "sampling/sampling_logp_difference/max": 1.8091429471969604, "sampling/sampling_logp_difference/mean": 0.030294643715023994, "step": 694 }, { "clip_ratio/high_max": 0.004360056365840137, "clip_ratio/high_mean": 0.004360056365840137, "clip_ratio/low_mean": 0.0062915480230003595, "clip_ratio/low_min": 0.0062915480230003595, "clip_ratio/region_mean": 0.010651604388840497, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 80.75, "completions/mean_terminated_length": 80.75, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.06276643788442016, "epoch": 0.02791500984054304, "frac_reward_zero_std": 0.0, "grad_norm": 3.7013967037200928, "learning_rate": 7.896969696969698e-06, "loss": -0.0124, "num_tokens": 1511199.0, "reward": 0.788793683052063, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9964468479156494, "reward_meter_std": 0.0019429969834163785, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.17156179249286652, "reward_total_composite_mean": 0.788793683052063, "reward_total_composite_std": 0.17156179249286652, "reward_total_mean": 0.788793683052063, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9964468479156494, "rewards/meter/std": 0.0019429969834163785, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.788793683052063, "rewards/total_composite/std": 0.17156179249286652, "sampling/importance_sampling_ratio/max": 1.8497363328933716, "sampling/importance_sampling_ratio/mean": 1.0024663209915161, "sampling/importance_sampling_ratio/min": 0.1775287538766861, "sampling/sampling_logp_difference/max": 1.7286226749420166, "sampling/sampling_logp_difference/mean": 0.012971381656825542, "step": 695 }, { "clip_ratio/high_max": 0.00876707280986011, "clip_ratio/high_mean": 0.00876707280986011, "clip_ratio/low_mean": 0.0020850637229159474, "clip_ratio/low_min": 0.0020850637229159474, "clip_ratio/region_mean": 0.010852136532776058, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 125.75, "completions/mean_terminated_length": 125.75, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.07676348416134715, "epoch": 0.027955175322327993, "frac_reward_zero_std": 0.0, "grad_norm": 3.119380235671997, "learning_rate": 7.893939393939395e-06, "loss": -0.0063, "num_tokens": 1513573.0, "reward": 0.6532999277114868, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9228010177612305, "reward_meter_std": 0.20124337077140808, "reward_repeat_penalty_mean": 0.7000000476837158, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.18960076570510864, "reward_total_composite_mean": 0.6532999277114868, "reward_total_composite_std": 0.18960076570510864, "reward_total_mean": 0.6532999277114868, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9228010177612305, "rewards/meter/std": 0.20124337077140808, "rewards/repeat_penalty/mean": 0.7000000476837158, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.6532999277114868, "rewards/total_composite/std": 0.18960076570510864, "sampling/importance_sampling_ratio/max": 1.9363112449645996, "sampling/importance_sampling_ratio/mean": 0.9984103441238403, "sampling/importance_sampling_ratio/min": 0.2494570016860962, "sampling/sampling_logp_difference/max": 1.3884687423706055, "sampling/sampling_logp_difference/mean": 0.016659488901495934, "step": 696 }, { "clip_ratio/high_max": 0.01933896285481751, "clip_ratio/high_mean": 0.01933896285481751, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/region_mean": 0.023185116704553366, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 59.375, "completions/mean_terminated_length": 59.375, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.09744885191321373, "epoch": 0.027995340804112947, "frac_reward_zero_std": 0.0, "grad_norm": 4.277256011962891, "learning_rate": 7.89090909090909e-06, "loss": 0.0606, "num_tokens": 1515360.0, "reward": 0.8894380331039429, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9654116630554199, "reward_meter_std": 0.034304678440093994, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.17405925691127777, "reward_total_composite_mean": 0.8894380331039429, "reward_total_composite_std": 0.17405925691127777, "reward_total_mean": 0.8894380331039429, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9654116630554199, "rewards/meter/std": 0.034304678440093994, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.8894380331039429, "rewards/total_composite/std": 0.17405925691127777, "sampling/importance_sampling_ratio/max": 1.4003076553344727, "sampling/importance_sampling_ratio/mean": 0.99860018491745, "sampling/importance_sampling_ratio/min": 0.3639322817325592, "sampling/sampling_logp_difference/max": 1.0107874870300293, "sampling/sampling_logp_difference/mean": 0.020010532811284065, "step": 697 }, { "clip_ratio/high_max": 0.015907755587249994, "clip_ratio/high_mean": 0.015907755587249994, "clip_ratio/low_mean": 0.007161757908761501, "clip_ratio/low_min": 0.007161757908761501, "clip_ratio/region_mean": 0.023069513496011496, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 412.0, "completions/mean_length": 382.25, "completions/mean_terminated_length": 363.71429443359375, "completions/min_length": 321.0, "completions/min_terminated_length": 321.0, "entropy": 0.18515709601342678, "epoch": 0.0280355062858979, "frac_reward_zero_std": 0.0, "grad_norm": 2.2967092990875244, "learning_rate": 7.88787878787879e-06, "loss": -0.0873, "num_tokens": 1519450.0, "reward": 0.18387337028980255, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8854166865348816, "reward_count_adherence_std": 0.1254950612783432, "reward_meter_mean": 0.6105729937553406, "reward_meter_std": 0.4790833294391632, "reward_repeat_penalty_mean": 0.3964124917984009, "reward_repeat_penalty_std": 0.25513792037963867, "reward_std": 0.21844197809696198, "reward_total_composite_mean": 0.18387337028980255, "reward_total_composite_std": 0.21844197809696198, "reward_total_mean": 0.18387337028980255, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8854166865348816, "rewards/count_adherence/std": 0.1254950612783432, "rewards/meter/mean": 0.6105729937553406, "rewards/meter/std": 0.4790833294391632, "rewards/repeat_penalty/mean": 0.3964124917984009, "rewards/repeat_penalty/std": 0.25513792037963867, "rewards/total_composite/mean": 0.18387337028980255, "rewards/total_composite/std": 0.21844197809696198, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9987672567367554, "sampling/importance_sampling_ratio/min": 0.001238961354829371, "sampling/sampling_logp_difference/max": 6.693481922149658, "sampling/sampling_logp_difference/mean": 0.0396236851811409, "step": 698 }, { "clip_ratio/high_max": 0.02036348171532154, "clip_ratio/high_mean": 0.02036348171532154, "clip_ratio/low_mean": 0.00747219193726778, "clip_ratio/low_min": 0.00747219193726778, "clip_ratio/region_mean": 0.02783567365258932, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.14588068891316652, "epoch": 0.028075671767682855, "frac_reward_zero_std": 0.0, "grad_norm": 8.473615646362305, "learning_rate": 7.884848484848485e-06, "loss": 0.0056, "num_tokens": 1521331.0, "reward": 0.8282062411308289, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9927859306335449, "reward_meter_std": 0.014541360549628735, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_std": 0.18183691799640656, "reward_total_composite_mean": 0.8282062411308289, "reward_total_composite_std": 0.18183693289756775, "reward_total_mean": 0.8282062411308289, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9927859306335449, "rewards/meter/std": 0.014541360549628735, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.8282062411308289, "rewards/total_composite/std": 0.18183693289756775, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9997567534446716, "sampling/importance_sampling_ratio/min": 0.1795266717672348, "sampling/sampling_logp_difference/max": 1.766160249710083, "sampling/sampling_logp_difference/mean": 0.03374454379081726, "step": 699 }, { "clip_ratio/high_max": 0.006157157360576093, "clip_ratio/high_mean": 0.006157157360576093, "clip_ratio/low_mean": 0.00364298140630126, "clip_ratio/low_min": 0.00364298140630126, "clip_ratio/region_mean": 0.009800138766877353, "completions/clipped_ratio": 0.0, "completions/max_length": 444.0, "completions/max_terminated_length": 444.0, "completions/mean_length": 414.875, "completions/mean_terminated_length": 414.875, "completions/min_length": 401.0, "completions/min_terminated_length": 401.0, "entropy": 0.045434954110533, "epoch": 0.02811583724946781, "frac_reward_zero_std": 0.0, "grad_norm": 0.8298696279525757, "learning_rate": 7.881818181818182e-06, "loss": 0.0097, "num_tokens": 1526242.0, "reward": 0.16019713878631592, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.029462797567248344, "reward_meter_mean": 0.9960967302322388, "reward_meter_std": 0.0013218529056757689, "reward_repeat_penalty_mean": 0.19121241569519043, "reward_repeat_penalty_std": 0.07494427263736725, "reward_std": 0.061176449060440063, "reward_total_composite_mean": 0.16019713878631592, "reward_total_composite_std": 0.061176449060440063, "reward_total_mean": 0.16019713878631592, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.029462797567248344, "rewards/meter/mean": 0.9960967302322388, "rewards/meter/std": 0.0013218529056757689, "rewards/repeat_penalty/mean": 0.19121241569519043, "rewards/repeat_penalty/std": 0.07494427263736725, "rewards/total_composite/mean": 0.16019713878631592, "rewards/total_composite/std": 0.061176449060440063, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0012154579162598, "sampling/importance_sampling_ratio/min": 0.238611102104187, "sampling/sampling_logp_difference/max": 1.432920217514038, "sampling/sampling_logp_difference/mean": 0.008506695739924908, "step": 700 }, { "epoch": 0.02811583724946781, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.11538461538461539, "eval_completions/max_length": 471.9230769230769, "eval_completions/max_terminated_length": 415.15384615384613, "eval_completions/mean_length": 249.43269230769232, "eval_completions/mean_terminated_length": 217.6739994929387, "eval_completions/min_length": 68.53846153846153, "eval_completions/min_terminated_length": 68.53846153846153, "eval_entropy": 0.1376804428604933, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 1526242.0, "eval_reward": 0.29438196466519284, "eval_reward_arabic_clean_mean": 0.9038461538461539, "eval_reward_arabic_clean_std": 0.2343954168833219, "eval_reward_count_adherence_mean": 0.8881359283740704, "eval_reward_count_adherence_std": 0.16248861929545036, "eval_reward_meter_mean": 0.6321157022164419, "eval_reward_meter_std": 0.37711624113413006, "eval_reward_repeat_penalty_mean": 0.5953293947073129, "eval_reward_repeat_penalty_std": 0.32899803152451146, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.29438196466519284, "eval_reward_total_composite_std": 0.28008361991781455, "eval_reward_total_mean": 0.29438196466519284, "eval_rewards/arabic_clean/mean": 0.9038461538461539, "eval_rewards/arabic_clean/std": 0.2343954168833219, "eval_rewards/count_adherence/mean": 0.8881359283740704, "eval_rewards/count_adherence/std": 0.16248861929545036, "eval_rewards/meter/mean": 0.6321157022164419, "eval_rewards/meter/std": 0.37711624113413006, "eval_rewards/repeat_penalty/mean": 0.5953293947073129, "eval_rewards/repeat_penalty/std": 0.32899803152451146, "eval_rewards/total_composite/mean": 0.29438196466519284, "eval_rewards/total_composite/std": 0.28008361991781455, "eval_runtime": 87.7418, "eval_samples_per_second": 1.185, "eval_sampling/importance_sampling_ratio/max": 1.455959943624643, "eval_sampling/importance_sampling_ratio/mean": 1.0041690973135142, "eval_sampling/importance_sampling_ratio/min": 0.4343368663237645, "eval_sampling/sampling_logp_difference/max": 0.881988103573139, "eval_sampling/sampling_logp_difference/mean": 0.012457216982371531, "eval_steps_per_second": 0.148, "step": 700 }, { "clip_ratio/high_max": 0.015958538744598627, "clip_ratio/high_mean": 0.015958538744598627, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.015958538744598627, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 138.75, "completions/mean_terminated_length": 85.42857360839844, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.11331802047789097, "epoch": 0.028156002731252763, "frac_reward_zero_std": 0.0, "grad_norm": 2.5312631130218506, "learning_rate": 7.87878787878788e-06, "loss": -0.12, "num_tokens": 1528168.0, "reward": 0.6387161016464233, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.646858811378479, "reward_meter_std": 0.40135496854782104, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.41526278853416443, "reward_total_composite_mean": 0.6387161016464233, "reward_total_composite_std": 0.41526278853416443, "reward_total_mean": 0.6387161016464233, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.646858811378479, "rewards/meter/std": 0.40135496854782104, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6387161016464233, "rewards/total_composite/std": 0.41526278853416443, "sampling/importance_sampling_ratio/max": 1.6849335432052612, "sampling/importance_sampling_ratio/mean": 1.0036530494689941, "sampling/importance_sampling_ratio/min": 0.5280055403709412, "sampling/sampling_logp_difference/max": 0.638648509979248, "sampling/sampling_logp_difference/mean": 0.017966095358133316, "step": 701 }, { "clip_ratio/high_max": 0.0155344782397151, "clip_ratio/high_mean": 0.0155344782397151, "clip_ratio/low_mean": 0.01317842910066247, "clip_ratio/low_min": 0.01317842910066247, "clip_ratio/region_mean": 0.02871290734037757, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 128.25, "completions/mean_terminated_length": 73.42857360839844, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.37723080068826675, "epoch": 0.028196168213037717, "frac_reward_zero_std": 0.0, "grad_norm": 3.0773770809173584, "learning_rate": 7.875757575757577e-06, "loss": -0.0691, "num_tokens": 1529770.0, "reward": 0.33759811520576477, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3803873062133789, "reward_meter_std": 0.26946189999580383, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3016302287578583, "reward_total_composite_mean": 0.33759811520576477, "reward_total_composite_std": 0.30163025856018066, "reward_total_mean": 0.33759811520576477, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3803873062133789, "rewards/meter/std": 0.26946189999580383, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.33759811520576477, "rewards/total_composite/std": 0.30163025856018066, "sampling/importance_sampling_ratio/max": 1.7729606628417969, "sampling/importance_sampling_ratio/mean": 1.011560082435608, "sampling/importance_sampling_ratio/min": 0.14418688416481018, "sampling/sampling_logp_difference/max": 1.9366450309753418, "sampling/sampling_logp_difference/mean": 0.04749147966504097, "step": 702 }, { "clip_ratio/high_max": 0.004737977171316743, "clip_ratio/high_mean": 0.004737977171316743, "clip_ratio/low_mean": 0.0028346364269964397, "clip_ratio/low_min": 0.0028346364269964397, "clip_ratio/region_mean": 0.007572613598313183, "completions/clipped_ratio": 0.0, "completions/max_length": 242.0, "completions/max_terminated_length": 242.0, "completions/mean_length": 229.75, "completions/mean_terminated_length": 229.75, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.03166929807048291, "epoch": 0.02823633369482267, "frac_reward_zero_std": 0.0, "grad_norm": 1.2964930534362793, "learning_rate": 7.872727272727273e-06, "loss": -0.0245, "num_tokens": 1533320.0, "reward": 0.2607119679450989, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970081448554993, "reward_meter_std": 0.001187506248243153, "reward_repeat_penalty_mean": 0.2613636255264282, "reward_repeat_penalty_std": 0.197011336684227, "reward_std": 0.1968017816543579, "reward_total_composite_mean": 0.2607119679450989, "reward_total_composite_std": 0.19680176675319672, "reward_total_mean": 0.2607119679450989, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970081448554993, "rewards/meter/std": 0.001187506248243153, "rewards/repeat_penalty/mean": 0.2613636255264282, "rewards/repeat_penalty/std": 0.197011336684227, "rewards/total_composite/mean": 0.2607119679450989, "rewards/total_composite/std": 0.19680176675319672, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9988393187522888, "sampling/importance_sampling_ratio/min": 0.009685891680419445, "sampling/sampling_logp_difference/max": 4.6370849609375, "sampling/sampling_logp_difference/mean": 0.014087834395468235, "step": 703 }, { "clip_ratio/high_max": 0.009915342554450035, "clip_ratio/high_mean": 0.009915342554450035, "clip_ratio/low_mean": 0.01385743310675025, "clip_ratio/low_min": 0.01385743310675025, "clip_ratio/region_mean": 0.023772775661200285, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 364.5, "completions/mean_terminated_length": 315.3333435058594, "completions/min_length": 229.0, "completions/min_terminated_length": 229.0, "entropy": 0.2017015889286995, "epoch": 0.028276499176607624, "frac_reward_zero_std": 0.0, "grad_norm": 3.015087127685547, "learning_rate": 7.86969696969697e-06, "loss": -0.1074, "num_tokens": 1536668.0, "reward": 0.0969100296497345, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8522727489471436, "reward_count_adherence_std": 0.21697448194026947, "reward_meter_mean": 0.20065514743328094, "reward_meter_std": 0.3053584098815918, "reward_repeat_penalty_mean": 0.6299689412117004, "reward_repeat_penalty_std": 0.28235092759132385, "reward_std": 0.21232594549655914, "reward_total_composite_mean": 0.0969100296497345, "reward_total_composite_std": 0.21232596039772034, "reward_total_mean": 0.0969100296497345, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8522727489471436, "rewards/count_adherence/std": 0.21697448194026947, "rewards/meter/mean": 0.20065514743328094, "rewards/meter/std": 0.3053584098815918, "rewards/repeat_penalty/mean": 0.6299689412117004, "rewards/repeat_penalty/std": 0.28235092759132385, "rewards/total_composite/mean": 0.0969100296497345, "rewards/total_composite/std": 0.21232596039772034, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9969695806503296, "sampling/importance_sampling_ratio/min": 0.08501863479614258, "sampling/sampling_logp_difference/max": 2.4648847579956055, "sampling/sampling_logp_difference/mean": 0.04316805675625801, "step": 704 }, { "clip_ratio/high_max": 0.006355932215228677, "clip_ratio/high_mean": 0.006355932215228677, "clip_ratio/low_mean": 0.006398809840902686, "clip_ratio/low_min": 0.006398809840902686, "clip_ratio/region_mean": 0.012754742056131363, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 58.75, "completions/mean_terminated_length": 58.75, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.053199955029413104, "epoch": 0.02831666465839258, "frac_reward_zero_std": 0.0, "grad_norm": 2.615363836288452, "learning_rate": 7.866666666666667e-06, "loss": -0.0075, "num_tokens": 1538402.0, "reward": 0.909184455871582, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9921605587005615, "reward_meter_std": 0.006017809733748436, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.15155236423015594, "reward_total_composite_mean": 0.909184455871582, "reward_total_composite_std": 0.15155236423015594, "reward_total_mean": 0.909184455871582, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9921605587005615, "rewards/meter/std": 0.006017809733748436, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.909184455871582, "rewards/total_composite/std": 0.15155236423015594, "sampling/importance_sampling_ratio/max": 1.5749088525772095, "sampling/importance_sampling_ratio/mean": 0.9994084239006042, "sampling/importance_sampling_ratio/min": 0.01880229450762272, "sampling/sampling_logp_difference/max": 3.973776340484619, "sampling/sampling_logp_difference/mean": 0.021948276087641716, "step": 705 }, { "clip_ratio/high_max": 0.01414728700183332, "clip_ratio/high_mean": 0.01414728700183332, "clip_ratio/low_mean": 0.0027777778450399637, "clip_ratio/low_min": 0.0027777778450399637, "clip_ratio/region_mean": 0.016925064846873283, "completions/clipped_ratio": 0.0, "completions/max_length": 45.0, "completions/max_terminated_length": 45.0, "completions/mean_length": 44.75, "completions/mean_terminated_length": 44.75, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.13661748263984919, "epoch": 0.028356830140177532, "frac_reward_zero_std": 0.0, "grad_norm": 4.515374183654785, "learning_rate": 7.863636363636364e-06, "loss": 0.0093, "num_tokens": 1539816.0, "reward": 0.9157871603965759, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9157871603965759, "reward_meter_std": 0.05085242539644241, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05085243284702301, "reward_total_composite_mean": 0.9157871603965759, "reward_total_composite_std": 0.05085242539644241, "reward_total_mean": 0.9157871603965759, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9157871603965759, "rewards/meter/std": 0.05085242539644241, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9157871603965759, "rewards/total_composite/std": 0.05085242539644241, "sampling/importance_sampling_ratio/max": 1.2550123929977417, "sampling/importance_sampling_ratio/mean": 0.997020423412323, "sampling/importance_sampling_ratio/min": 0.4525808095932007, "sampling/sampling_logp_difference/max": 0.7927889823913574, "sampling/sampling_logp_difference/mean": 0.02820507250726223, "step": 706 }, { "clip_ratio/high_max": 0.013377037481404841, "clip_ratio/high_mean": 0.013377037481404841, "clip_ratio/low_mean": 0.007405462441965938, "clip_ratio/low_min": 0.007405462441965938, "clip_ratio/region_mean": 0.02078249992337078, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 84.625, "completions/mean_terminated_length": 84.625, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.15222040470689535, "epoch": 0.028396995621962486, "frac_reward_zero_std": 0.0, "grad_norm": 2.2196364402770996, "learning_rate": 7.860606060606062e-06, "loss": -0.0003, "num_tokens": 1541901.0, "reward": 0.9956607818603516, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956607818603516, "reward_meter_std": 0.0007934165187180042, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007934237364679575, "reward_total_composite_mean": 0.9956607818603516, "reward_total_composite_std": 0.0007934165187180042, "reward_total_mean": 0.9956607818603516, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956607818603516, "rewards/meter/std": 0.0007934165187180042, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956607818603516, "rewards/total_composite/std": 0.0007934165187180042, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0091147422790527, "sampling/importance_sampling_ratio/min": 0.4566366374492645, "sampling/sampling_logp_difference/max": 0.8360247611999512, "sampling/sampling_logp_difference/mean": 0.026080820709466934, "step": 707 }, { "clip_ratio/high_max": 0.010587403550744057, "clip_ratio/high_mean": 0.010587403550744057, "clip_ratio/low_mean": 0.012675938894972205, "clip_ratio/low_min": 0.012675938894972205, "clip_ratio/region_mean": 0.023263342445716262, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 330.75, "completions/mean_terminated_length": 330.75, "completions/min_length": 284.0, "completions/min_terminated_length": 284.0, "entropy": 0.10461874585598707, "epoch": 0.02843716110374744, "frac_reward_zero_std": 0.0, "grad_norm": 2.581402063369751, "learning_rate": 7.857575757575759e-06, "loss": 0.0303, "num_tokens": 1546099.0, "reward": 0.2035118043422699, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9027777910232544, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.5674466490745544, "reward_meter_std": 0.4153376519680023, "reward_repeat_penalty_mean": 0.34049707651138306, "reward_repeat_penalty_std": 0.22640277445316315, "reward_std": 0.23429378867149353, "reward_total_composite_mean": 0.2035118043422699, "reward_total_composite_std": 0.23429380357265472, "reward_total_mean": 0.2035118043422699, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9027777910232544, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.5674466490745544, "rewards/meter/std": 0.4153376519680023, "rewards/repeat_penalty/mean": 0.34049707651138306, "rewards/repeat_penalty/std": 0.22640277445316315, "rewards/total_composite/mean": 0.2035118043422699, "rewards/total_composite/std": 0.23429380357265472, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002025842666626, "sampling/importance_sampling_ratio/min": 0.026643957942724228, "sampling/sampling_logp_difference/max": 3.625192880630493, "sampling/sampling_logp_difference/mean": 0.023940538987517357, "step": 708 }, { "clip_ratio/high_max": 0.005005840037483722, "clip_ratio/high_mean": 0.005005840037483722, "clip_ratio/low_mean": 0.0015743073308840394, "clip_ratio/low_min": 0.0015743073308840394, "clip_ratio/region_mean": 0.006580147368367761, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 452.25, "completions/mean_terminated_length": 416.3999938964844, "completions/min_length": 372.0, "completions/min_terminated_length": 372.0, "entropy": 0.0702343238517642, "epoch": 0.028477326585532394, "frac_reward_zero_std": 0.0, "grad_norm": 0.8571996688842773, "learning_rate": 7.854545454545454e-06, "loss": -0.3582, "num_tokens": 1549805.0, "reward": 0.25732704997062683, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.6590909361839294, "reward_count_adherence_std": 0.39101481437683105, "reward_meter_mean": 0.8675932884216309, "reward_meter_std": 0.35074320435523987, "reward_repeat_penalty_mean": 0.6495236158370972, "reward_repeat_penalty_std": 0.30929210782051086, "reward_std": 0.2531158924102783, "reward_total_composite_mean": 0.25732704997062683, "reward_total_composite_std": 0.2531158924102783, "reward_total_mean": 0.25732704997062683, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.6590909361839294, "rewards/count_adherence/std": 0.39101481437683105, "rewards/meter/mean": 0.8675932884216309, "rewards/meter/std": 0.35074320435523987, "rewards/repeat_penalty/mean": 0.6495236158370972, "rewards/repeat_penalty/std": 0.30929210782051086, "rewards/total_composite/mean": 0.25732704997062683, "rewards/total_composite/std": 0.2531158924102783, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0035980939865112, "sampling/importance_sampling_ratio/min": 0.05962793529033661, "sampling/sampling_logp_difference/max": 2.8196310997009277, "sampling/sampling_logp_difference/mean": 0.019949674606323242, "step": 709 }, { "clip_ratio/high_max": 0.004514876694884151, "clip_ratio/high_mean": 0.004514876694884151, "clip_ratio/low_mean": 0.0021170872496441007, "clip_ratio/low_min": 0.0021170872496441007, "clip_ratio/region_mean": 0.006631963944528252, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 474.875, "completions/mean_terminated_length": 469.5714416503906, "completions/min_length": 439.0, "completions/min_terminated_length": 439.0, "entropy": 0.05670151812955737, "epoch": 0.028517492067317348, "frac_reward_zero_std": 0.0, "grad_norm": 0.6216875314712524, "learning_rate": 7.851515151515152e-06, "loss": -0.2247, "num_tokens": 1554988.0, "reward": 0.3333708643913269, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.7767857313156128, "reward_count_adherence_std": 0.20360276103019714, "reward_meter_mean": 0.9897267818450928, "reward_meter_std": 0.019744135439395905, "reward_repeat_penalty_mean": 0.5224603414535522, "reward_repeat_penalty_std": 0.23697951436042786, "reward_std": 0.18023891746997833, "reward_total_composite_mean": 0.3333708643913269, "reward_total_composite_std": 0.18023891746997833, "reward_total_mean": 0.3333708643913269, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.7767857313156128, "rewards/count_adherence/std": 0.20360276103019714, "rewards/meter/mean": 0.9897267818450928, "rewards/meter/std": 0.019744135439395905, "rewards/repeat_penalty/mean": 0.5224603414535522, "rewards/repeat_penalty/std": 0.23697951436042786, "rewards/total_composite/mean": 0.3333708643913269, "rewards/total_composite/std": 0.18023891746997833, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017552375793457, "sampling/importance_sampling_ratio/min": 0.08268983662128448, "sampling/sampling_logp_difference/max": 2.4926586151123047, "sampling/sampling_logp_difference/mean": 0.010584630072116852, "step": 710 }, { "clip_ratio/high_max": 0.0016890882980078459, "clip_ratio/high_mean": 0.0016890882980078459, "clip_ratio/low_mean": 0.0008561643480788916, "clip_ratio/low_min": 0.0008561643480788916, "clip_ratio/region_mean": 0.0025452526460867375, "completions/clipped_ratio": 0.0, "completions/max_length": 461.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 440.25, "completions/mean_terminated_length": 440.25, "completions/min_length": 436.0, "completions/min_terminated_length": 436.0, "entropy": 0.011412000167183578, "epoch": 0.028557657549102302, "frac_reward_zero_std": 0.0, "grad_norm": 0.6046677231788635, "learning_rate": 7.848484848484849e-06, "loss": -0.0049, "num_tokens": 1560550.0, "reward": 0.053851306438446045, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7946428060531616, "reward_count_adherence_std": 0.025253823027014732, "reward_meter_mean": 0.9979845285415649, "reward_meter_std": 0.00018585202633403242, "reward_repeat_penalty_mean": 0.0676877498626709, "reward_repeat_penalty_std": 0.023803479969501495, "reward_std": 0.019493909552693367, "reward_total_composite_mean": 0.053851306438446045, "reward_total_composite_std": 0.019493911415338516, "reward_total_mean": 0.053851306438446045, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7946428060531616, "rewards/count_adherence/std": 0.025253823027014732, "rewards/meter/mean": 0.9979845285415649, "rewards/meter/std": 0.00018585202633403242, "rewards/repeat_penalty/mean": 0.0676877498626709, "rewards/repeat_penalty/std": 0.023803479969501495, "rewards/total_composite/mean": 0.053851306438446045, "rewards/total_composite/std": 0.019493911415338516, "sampling/importance_sampling_ratio/max": 1.460545539855957, "sampling/importance_sampling_ratio/mean": 1.0001322031021118, "sampling/importance_sampling_ratio/min": 0.30463549494743347, "sampling/sampling_logp_difference/max": 1.1886392831802368, "sampling/sampling_logp_difference/mean": 0.003577793249860406, "step": 711 }, { "clip_ratio/high_max": 0.014235412469133735, "clip_ratio/high_mean": 0.014235412469133735, "clip_ratio/low_mean": 0.016306631732732058, "clip_ratio/low_min": 0.016306631732732058, "clip_ratio/region_mean": 0.030542044201865792, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.5, "completions/mean_terminated_length": 69.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.1954718241468072, "epoch": 0.028597823030887256, "frac_reward_zero_std": 0.0, "grad_norm": 4.805282115936279, "learning_rate": 7.845454545454546e-06, "loss": 0.0004, "num_tokens": 1562410.0, "reward": 0.7055873274803162, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8851493000984192, "reward_meter_std": 0.1646534502506256, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.216535747051239, "reward_total_composite_mean": 0.7055873274803162, "reward_total_composite_std": 0.21653573215007782, "reward_total_mean": 0.7055873274803162, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8851493000984192, "rewards/meter/std": 0.1646534502506256, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7055873274803162, "rewards/total_composite/std": 0.21653573215007782, "sampling/importance_sampling_ratio/max": 1.8839155435562134, "sampling/importance_sampling_ratio/mean": 0.997905969619751, "sampling/importance_sampling_ratio/min": 0.1394468992948532, "sampling/sampling_logp_difference/max": 1.9700713157653809, "sampling/sampling_logp_difference/mean": 0.03851144388318062, "step": 712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.02863798851267221, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 7.842424242424243e-06, "loss": 0.0, "num_tokens": 1564234.0, "reward": 0.4566306471824646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8166667222976685, "reward_count_adherence_std": 0.030860668048262596, "reward_meter_mean": 0.9704335927963257, "reward_meter_std": 0.07359233498573303, "reward_repeat_penalty_mean": 0.5789903998374939, "reward_repeat_penalty_std": 0.045423366129398346, "reward_std": 0.02297401800751686, "reward_total_composite_mean": 0.4566306471824646, "reward_total_composite_std": 0.02297402359545231, "reward_total_mean": 0.4566306471824646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8166667222976685, "rewards/count_adherence/std": 0.030860668048262596, "rewards/meter/mean": 0.9704335927963257, "rewards/meter/std": 0.07359233498573303, "rewards/repeat_penalty/mean": 0.5789903998374939, "rewards/repeat_penalty/std": 0.045423366129398346, "rewards/total_composite/mean": 0.4566306471824646, "rewards/total_composite/std": 0.02297402359545231, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 713 }, { "clip_ratio/high_max": 0.0012019231216982007, "clip_ratio/high_mean": 0.0012019231216982007, "clip_ratio/low_mean": 0.0018522579048294574, "clip_ratio/low_min": 0.0018522579048294574, "clip_ratio/region_mean": 0.003054181026527658, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 336.5, "completions/mean_terminated_length": 336.5, "completions/min_length": 311.0, "completions/min_terminated_length": 311.0, "entropy": 0.018028545891866088, "epoch": 0.028678153994457164, "frac_reward_zero_std": 0.0, "grad_norm": 0.7193350791931152, "learning_rate": 7.83939393939394e-06, "loss": 0.0251, "num_tokens": 1568398.0, "reward": 0.13619501888751984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977849721908569, "reward_meter_std": 0.0006831432110629976, "reward_repeat_penalty_mean": 0.19117647409439087, "reward_repeat_penalty_std": 0.16262187063694, "reward_std": 0.11569356918334961, "reward_total_composite_mean": 0.13619501888751984, "reward_total_composite_std": 0.1156935766339302, "reward_total_mean": 0.13619501888751984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977849721908569, "rewards/meter/std": 0.0006831432110629976, "rewards/repeat_penalty/mean": 0.19117647409439087, "rewards/repeat_penalty/std": 0.16262187063694, "rewards/total_composite/mean": 0.13619501888751984, "rewards/total_composite/std": 0.1156935766339302, "sampling/importance_sampling_ratio/max": 1.3743659257888794, "sampling/importance_sampling_ratio/mean": 0.9997386932373047, "sampling/importance_sampling_ratio/min": 0.2750125825405121, "sampling/sampling_logp_difference/max": 1.290938377380371, "sampling/sampling_logp_difference/mean": 0.003576630027964711, "step": 714 }, { "clip_ratio/high_max": 0.018382353708148003, "clip_ratio/high_mean": 0.018382353708148003, "clip_ratio/low_mean": 0.023657660058233887, "clip_ratio/low_min": 0.023657660058233887, "clip_ratio/region_mean": 0.04204001376638189, "completions/clipped_ratio": 0.0, "completions/max_length": 131.0, "completions/max_terminated_length": 131.0, "completions/mean_length": 123.625, "completions/mean_terminated_length": 123.625, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.23239293694496155, "epoch": 0.028718319476242118, "frac_reward_zero_std": 0.0, "grad_norm": 4.87160062789917, "learning_rate": 7.836363636363638e-06, "loss": 0.0035, "num_tokens": 1570787.0, "reward": 0.7246866822242737, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9901050329208374, "reward_meter_std": 0.0037112515419721603, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.1608559489250183, "reward_std": 0.15871338546276093, "reward_total_composite_mean": 0.7246866822242737, "reward_total_composite_std": 0.15871340036392212, "reward_total_mean": 0.7246866822242737, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9901050329208374, "rewards/meter/std": 0.0037112515419721603, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.1608559489250183, "rewards/total_composite/mean": 0.7246866822242737, "rewards/total_composite/std": 0.15871340036392212, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9998616576194763, "sampling/importance_sampling_ratio/min": 0.19606797397136688, "sampling/sampling_logp_difference/max": 1.6292939186096191, "sampling/sampling_logp_difference/mean": 0.04571465030312538, "step": 715 }, { "clip_ratio/high_max": 0.0044064579415135086, "clip_ratio/high_mean": 0.0044064579415135086, "clip_ratio/low_mean": 0.010836752247996628, "clip_ratio/low_min": 0.010836752247996628, "clip_ratio/region_mean": 0.015243210189510137, "completions/clipped_ratio": 0.0, "completions/max_length": 215.0, "completions/max_terminated_length": 215.0, "completions/mean_length": 175.75, "completions/mean_terminated_length": 175.75, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.15365072712302208, "epoch": 0.02875848495802707, "frac_reward_zero_std": 0.0, "grad_norm": 2.7540860176086426, "learning_rate": 7.833333333333333e-06, "loss": 0.0175, "num_tokens": 1573577.0, "reward": 0.4638897478580475, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.8235251903533936, "reward_meter_std": 0.28830382227897644, "reward_repeat_penalty_mean": 0.6091269850730896, "reward_repeat_penalty_std": 0.19641855359077454, "reward_std": 0.2275657057762146, "reward_total_composite_mean": 0.4638897478580475, "reward_total_composite_std": 0.2275657206773758, "reward_total_mean": 0.4638897478580475, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.8235251903533936, "rewards/meter/std": 0.28830382227897644, "rewards/repeat_penalty/mean": 0.6091269850730896, "rewards/repeat_penalty/std": 0.19641855359077454, "rewards/total_composite/mean": 0.4638897478580475, "rewards/total_composite/std": 0.2275657206773758, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0048706531524658, "sampling/importance_sampling_ratio/min": 0.09449966251850128, "sampling/sampling_logp_difference/max": 2.359158992767334, "sampling/sampling_logp_difference/mean": 0.02091868966817856, "step": 716 }, { "clip_ratio/high_max": 0.010442280676215887, "clip_ratio/high_mean": 0.010442280676215887, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.010442280676215887, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 210.0, "completions/mean_length": 391.0, "completions/mean_terminated_length": 189.33334350585938, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.13471947237849236, "epoch": 0.028798650439812026, "frac_reward_zero_std": 0.0, "grad_norm": 1.3285834789276123, "learning_rate": 7.83030303030303e-06, "loss": -0.1657, "num_tokens": 1575801.0, "reward": 0.08441510796546936, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_count_adherence_mean": 0.824999988079071, "reward_count_adherence_std": 0.12817399203777313, "reward_meter_mean": 0.3074490427970886, "reward_meter_std": 0.2025083303451538, "reward_repeat_penalty_mean": 0.8395833373069763, "reward_repeat_penalty_std": 0.25571832060813904, "reward_std": 0.10946667194366455, "reward_total_composite_mean": 0.08441510796546936, "reward_total_composite_std": 0.10946666449308395, "reward_total_mean": 0.08441510796546936, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/count_adherence/mean": 0.824999988079071, "rewards/count_adherence/std": 0.12817399203777313, "rewards/meter/mean": 0.3074490427970886, "rewards/meter/std": 0.2025083303451538, "rewards/repeat_penalty/mean": 0.8395833373069763, "rewards/repeat_penalty/std": 0.25571832060813904, "rewards/total_composite/mean": 0.08441510796546936, "rewards/total_composite/std": 0.10946666449308395, "sampling/importance_sampling_ratio/max": 1.6081585884094238, "sampling/importance_sampling_ratio/mean": 1.002199649810791, "sampling/importance_sampling_ratio/min": 0.0001644995791139081, "sampling/sampling_logp_difference/max": 8.712602615356445, "sampling/sampling_logp_difference/mean": 0.05243195965886116, "step": 717 }, { "clip_ratio/high_max": 0.023710741428658366, "clip_ratio/high_mean": 0.023710741428658366, "clip_ratio/low_mean": 0.006465517450124025, "clip_ratio/low_min": 0.006465517450124025, "clip_ratio/region_mean": 0.03017625887878239, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 57.75, "completions/mean_terminated_length": 57.75, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.14472659677267075, "epoch": 0.02883881592159698, "frac_reward_zero_std": 0.0, "grad_norm": 6.637593746185303, "learning_rate": 7.827272727272728e-06, "loss": 0.0058, "num_tokens": 1577519.0, "reward": 0.9619162082672119, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9619162082672119, "reward_meter_std": 0.06612562388181686, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06612562388181686, "reward_total_composite_mean": 0.9619162082672119, "reward_total_composite_std": 0.06612562388181686, "reward_total_mean": 0.9619162082672119, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9619162082672119, "rewards/meter/std": 0.06612562388181686, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9619162082672119, "rewards/total_composite/std": 0.06612562388181686, "sampling/importance_sampling_ratio/max": 1.769487977027893, "sampling/importance_sampling_ratio/mean": 1.004167914390564, "sampling/importance_sampling_ratio/min": 0.20535492897033691, "sampling/sampling_logp_difference/max": 1.5830154418945312, "sampling/sampling_logp_difference/mean": 0.03136257454752922, "step": 718 }, { "clip_ratio/high_max": 0.005198180675506592, "clip_ratio/high_mean": 0.005198180675506592, "clip_ratio/low_mean": 0.0008802816737443209, "clip_ratio/low_min": 0.0008802816737443209, "clip_ratio/region_mean": 0.006078462349250913, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 171.0, "completions/mean_length": 208.875, "completions/mean_terminated_length": 165.57144165039062, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.07858177460730076, "epoch": 0.028878981403381934, "frac_reward_zero_std": 0.0, "grad_norm": 0.982711136341095, "learning_rate": 7.824242424242425e-06, "loss": -0.1972, "num_tokens": 1580014.0, "reward": 0.36472243070602417, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8615807294845581, "reward_meter_std": 0.1756223738193512, "reward_repeat_penalty_mean": 0.5022321939468384, "reward_repeat_penalty_std": 0.19195686280727386, "reward_std": 0.1904788315296173, "reward_total_composite_mean": 0.36472243070602417, "reward_total_composite_std": 0.1904788464307785, "reward_total_mean": 0.36472243070602417, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8615807294845581, "rewards/meter/std": 0.1756223738193512, "rewards/repeat_penalty/mean": 0.5022321939468384, "rewards/repeat_penalty/std": 0.19195686280727386, "rewards/total_composite/mean": 0.36472243070602417, "rewards/total_composite/std": 0.1904788464307785, "sampling/importance_sampling_ratio/max": 1.6336287260055542, "sampling/importance_sampling_ratio/mean": 1.0052262544631958, "sampling/importance_sampling_ratio/min": 0.5830943584442139, "sampling/sampling_logp_difference/max": 0.5394062995910645, "sampling/sampling_logp_difference/mean": 0.011406843550503254, "step": 719 }, { "clip_ratio/high_max": 0.015587894711643457, "clip_ratio/high_mean": 0.015587894711643457, "clip_ratio/low_mean": 0.0012820513220503926, "clip_ratio/low_min": 0.0012820513220503926, "clip_ratio/region_mean": 0.01686994603369385, "completions/clipped_ratio": 0.0, "completions/max_length": 243.0, "completions/max_terminated_length": 243.0, "completions/mean_length": 228.625, "completions/mean_terminated_length": 228.625, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.08757321583107114, "epoch": 0.028919146885166887, "frac_reward_zero_std": 0.0, "grad_norm": 1.5212147235870361, "learning_rate": 7.821212121212122e-06, "loss": -0.0478, "num_tokens": 1583419.0, "reward": 0.3818710744380951, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.06613000482320786, "reward_meter_mean": 0.9813181161880493, "reward_meter_std": 0.009512215852737427, "reward_repeat_penalty_mean": 0.527634859085083, "reward_repeat_penalty_std": 0.18670302629470825, "reward_std": 0.13756079971790314, "reward_total_composite_mean": 0.3818710744380951, "reward_total_composite_std": 0.13756081461906433, "reward_total_mean": 0.3818710744380951, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.06613000482320786, "rewards/meter/mean": 0.9813181161880493, "rewards/meter/std": 0.009512215852737427, "rewards/repeat_penalty/mean": 0.527634859085083, "rewards/repeat_penalty/std": 0.18670302629470825, "rewards/total_composite/mean": 0.3818710744380951, "rewards/total_composite/std": 0.13756081461906433, "sampling/importance_sampling_ratio/max": 1.7837820053100586, "sampling/importance_sampling_ratio/mean": 1.0003083944320679, "sampling/importance_sampling_ratio/min": 0.14308412373065948, "sampling/sampling_logp_difference/max": 1.9443225860595703, "sampling/sampling_logp_difference/mean": 0.014808963052928448, "step": 720 }, { "clip_ratio/high_max": 0.05381742771714926, "clip_ratio/high_mean": 0.05381742771714926, "clip_ratio/low_mean": 0.051155281253159046, "clip_ratio/low_min": 0.051155281253159046, "clip_ratio/region_mean": 0.1049727089703083, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.5916384495794773, "epoch": 0.02895931236695184, "frac_reward_zero_std": 0.0, "grad_norm": 9.25263786315918, "learning_rate": 7.81818181818182e-06, "loss": -0.0114, "num_tokens": 1585267.0, "reward": 0.5946333408355713, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6475868821144104, "reward_meter_std": 0.3421926498413086, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.34638410806655884, "reward_total_composite_mean": 0.5946333408355713, "reward_total_composite_std": 0.34638410806655884, "reward_total_mean": 0.5946333408355713, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6475868821144104, "rewards/meter/std": 0.3421926498413086, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.5946333408355713, "rewards/total_composite/std": 0.34638410806655884, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.991131603717804, "sampling/importance_sampling_ratio/min": 0.052791085094213486, "sampling/sampling_logp_difference/max": 2.941412925720215, "sampling/sampling_logp_difference/mean": 0.1203828752040863, "step": 721 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 92.0, "completions/max_terminated_length": 92.0, "completions/mean_length": 48.25, "completions/mean_terminated_length": 48.25, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.10689277853816748, "epoch": 0.028999477848736795, "frac_reward_zero_std": 0.0, "grad_norm": 7.886468410491943, "learning_rate": 7.815151515151515e-06, "loss": 0.3273, "num_tokens": 1586837.0, "reward": 0.8698122501373291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.9944906830787659, "reward_meter_std": 0.0011895333882421255, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35145723819732666, "reward_total_composite_mean": 0.8698122501373291, "reward_total_composite_std": 0.35145723819732666, "reward_total_mean": 0.8698122501373291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.9944906830787659, "rewards/meter/std": 0.0011895333882421255, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8698122501373291, "rewards/total_composite/std": 0.35145723819732666, "sampling/importance_sampling_ratio/max": 1.3288551568984985, "sampling/importance_sampling_ratio/mean": 1.0014286041259766, "sampling/importance_sampling_ratio/min": 0.4765561819076538, "sampling/sampling_logp_difference/max": 0.7411696910858154, "sampling/sampling_logp_difference/mean": 0.01547156646847725, "step": 722 }, { "clip_ratio/high_max": 0.0007826215587556362, "clip_ratio/high_mean": 0.0007826215587556362, "clip_ratio/low_mean": 0.0007961043156683445, "clip_ratio/low_min": 0.0007961043156683445, "clip_ratio/region_mean": 0.0015787258744239807, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 494.0, "completions/mean_length": 485.75, "completions/mean_terminated_length": 470.0, "completions/min_length": 452.0, "completions/min_terminated_length": 452.0, "entropy": 0.018016068963333964, "epoch": 0.02903964333052175, "frac_reward_zero_std": 0.0, "grad_norm": 0.8348656892776489, "learning_rate": 7.812121212121213e-06, "loss": -0.2014, "num_tokens": 1590899.0, "reward": 0.09887628257274628, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9464285373687744, "reward_count_adherence_std": 0.03306501731276512, "reward_meter_mean": 0.996362566947937, "reward_meter_std": 0.0027744658291339874, "reward_repeat_penalty_mean": 0.10470085591077805, "reward_repeat_penalty_std": 0.10408195108175278, "reward_std": 0.09698235988616943, "reward_total_composite_mean": 0.09887628257274628, "reward_total_composite_std": 0.09698235988616943, "reward_total_mean": 0.09887628257274628, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9464285373687744, "rewards/count_adherence/std": 0.03306501731276512, "rewards/meter/mean": 0.996362566947937, "rewards/meter/std": 0.0027744658291339874, "rewards/repeat_penalty/mean": 0.10470085591077805, "rewards/repeat_penalty/std": 0.10408195108175278, "rewards/total_composite/mean": 0.09887628257274628, "rewards/total_composite/std": 0.09698235988616943, "sampling/importance_sampling_ratio/max": 1.536428451538086, "sampling/importance_sampling_ratio/mean": 1.0003700256347656, "sampling/importance_sampling_ratio/min": 0.13657712936401367, "sampling/sampling_logp_difference/max": 1.990865707397461, "sampling/sampling_logp_difference/mean": 0.007093369495123625, "step": 723 }, { "clip_ratio/high_max": 0.002920488826930523, "clip_ratio/high_mean": 0.002920488826930523, "clip_ratio/low_mean": 0.0028556776233017445, "clip_ratio/low_min": 0.0028556776233017445, "clip_ratio/region_mean": 0.005776166450232267, "completions/clipped_ratio": 0.0, "completions/max_length": 403.0, "completions/max_terminated_length": 403.0, "completions/mean_length": 390.625, "completions/mean_terminated_length": 390.625, "completions/min_length": 379.0, "completions/min_terminated_length": 379.0, "entropy": 0.028950548847205937, "epoch": 0.029079808812306703, "frac_reward_zero_std": 0.0, "grad_norm": 1.2350369691848755, "learning_rate": 7.80909090909091e-06, "loss": 0.0117, "num_tokens": 1595768.0, "reward": 0.1255086362361908, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.734375, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9978216886520386, "reward_meter_std": 0.0007184247369877994, "reward_repeat_penalty_mean": 0.17863407731056213, "reward_repeat_penalty_std": 0.16910506784915924, "reward_std": 0.10828038305044174, "reward_total_composite_mean": 0.1255086362361908, "reward_total_composite_std": 0.10828039050102234, "reward_total_mean": 0.1255086362361908, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.734375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9978216886520386, "rewards/meter/std": 0.0007184247369877994, "rewards/repeat_penalty/mean": 0.17863407731056213, "rewards/repeat_penalty/std": 0.16910506784915924, "rewards/total_composite/mean": 0.1255086362361908, "rewards/total_composite/std": 0.10828039050102234, "sampling/importance_sampling_ratio/max": 1.8996793031692505, "sampling/importance_sampling_ratio/mean": 0.9993754625320435, "sampling/importance_sampling_ratio/min": 0.22015291452407837, "sampling/sampling_logp_difference/max": 1.5134329795837402, "sampling/sampling_logp_difference/mean": 0.0057006035931408405, "step": 724 }, { "clip_ratio/high_max": 0.007217321544885635, "clip_ratio/high_mean": 0.007217321544885635, "clip_ratio/low_mean": 0.004787406767718494, "clip_ratio/low_min": 0.004787406767718494, "clip_ratio/region_mean": 0.01200472831260413, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 136.75, "completions/mean_terminated_length": 83.14286041259766, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.09931796230375767, "epoch": 0.029119974294091657, "frac_reward_zero_std": 0.0, "grad_norm": 1.6505030393600464, "learning_rate": 7.806060606060607e-06, "loss": -0.1213, "num_tokens": 1597686.0, "reward": 0.5530991554260254, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5868990421295166, "reward_meter_std": 0.38611555099487305, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.4085244834423065, "reward_total_composite_mean": 0.5530991554260254, "reward_total_composite_std": 0.40852445363998413, "reward_total_mean": 0.5530991554260254, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5868990421295166, "rewards/meter/std": 0.38611555099487305, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.5530991554260254, "rewards/total_composite/std": 0.40852445363998413, "sampling/importance_sampling_ratio/max": 1.7016674280166626, "sampling/importance_sampling_ratio/mean": 1.0019892454147339, "sampling/importance_sampling_ratio/min": 0.4019929766654968, "sampling/sampling_logp_difference/max": 0.911320686340332, "sampling/sampling_logp_difference/mean": 0.022672384977340698, "step": 725 }, { "clip_ratio/high_max": 0.0009765625, "clip_ratio/high_mean": 0.0009765625, "clip_ratio/low_mean": 0.0029605674790218472, "clip_ratio/low_min": 0.0029605674790218472, "clip_ratio/region_mean": 0.003937129979021847, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 127.25, "completions/mean_terminated_length": 127.25, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.05762754753232002, "epoch": 0.02916013977587661, "frac_reward_zero_std": 0.0, "grad_norm": 1.0429009199142456, "learning_rate": 7.803030303030303e-06, "loss": -0.0023, "num_tokens": 1600008.0, "reward": 0.46501004695892334, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9302235841751099, "reward_meter_std": 0.015423719771206379, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.1511857956647873, "reward_std": 0.14159856736660004, "reward_total_composite_mean": 0.46501004695892334, "reward_total_composite_std": 0.14159858226776123, "reward_total_mean": 0.46501004695892334, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9302235841751099, "rewards/meter/std": 0.015423719771206379, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.1511857956647873, "rewards/total_composite/mean": 0.46501004695892334, "rewards/total_composite/std": 0.14159858226776123, "sampling/importance_sampling_ratio/max": 1.5183614492416382, "sampling/importance_sampling_ratio/mean": 1.002754807472229, "sampling/importance_sampling_ratio/min": 0.4002356231212616, "sampling/sampling_logp_difference/max": 0.9157018661499023, "sampling/sampling_logp_difference/mean": 0.01078812312334776, "step": 726 }, { "clip_ratio/high_max": 0.02533797360956669, "clip_ratio/high_mean": 0.02533797360956669, "clip_ratio/low_mean": 0.032587712397798896, "clip_ratio/low_min": 0.032587712397798896, "clip_ratio/region_mean": 0.057925686007365584, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 99.375, "completions/mean_terminated_length": 99.375, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.2684608269482851, "epoch": 0.029200305257661565, "frac_reward_zero_std": 0.0, "grad_norm": 7.388904094696045, "learning_rate": 7.800000000000002e-06, "loss": 0.0279, "num_tokens": 1601923.0, "reward": 0.3363860249519348, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4179952144622803, "reward_meter_std": 0.45823782682418823, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.36587125062942505, "reward_total_composite_mean": 0.3363860249519348, "reward_total_composite_std": 0.36587125062942505, "reward_total_mean": 0.3363860249519348, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4179952144622803, "rewards/meter/std": 0.45823782682418823, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.3363860249519348, "rewards/total_composite/std": 0.36587125062942505, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9850558042526245, "sampling/importance_sampling_ratio/min": 0.006547517143189907, "sampling/sampling_logp_difference/max": 5.028669357299805, "sampling/sampling_logp_difference/mean": 0.0847281664609909, "step": 727 }, { "clip_ratio/high_max": 0.003258531214669347, "clip_ratio/high_mean": 0.003258531214669347, "clip_ratio/low_mean": 0.008432107453700155, "clip_ratio/low_min": 0.008432107453700155, "clip_ratio/region_mean": 0.011690638668369502, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 459.125, "completions/mean_terminated_length": 459.125, "completions/min_length": 427.0, "completions/min_terminated_length": 427.0, "entropy": 0.06644301256164908, "epoch": 0.02924047073944652, "frac_reward_zero_std": 0.0, "grad_norm": 2.0805506706237793, "learning_rate": 7.796969696969697e-06, "loss": 0.0076, "num_tokens": 1607332.0, "reward": 0.12284211814403534, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0431290864944458, "reward_meter_mean": 0.28689688444137573, "reward_meter_std": 0.3819184899330139, "reward_repeat_penalty_mean": 0.5729086995124817, "reward_repeat_penalty_std": 0.018385794013738632, "reward_std": 0.1601785272359848, "reward_total_composite_mean": 0.12284211814403534, "reward_total_composite_std": 0.1601785272359848, "reward_total_mean": 0.12284211814403534, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0431290864944458, "rewards/meter/mean": 0.28689688444137573, "rewards/meter/std": 0.3819184899330139, "rewards/repeat_penalty/mean": 0.5729086995124817, "rewards/repeat_penalty/std": 0.018385794013738632, "rewards/total_composite/mean": 0.12284211814403534, "rewards/total_composite/std": 0.1601785272359848, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9993253350257874, "sampling/importance_sampling_ratio/min": 0.03411376476287842, "sampling/sampling_logp_difference/max": 3.37805438041687, "sampling/sampling_logp_difference/mean": 0.019973881542682648, "step": 728 }, { "clip_ratio/high_max": 0.007218867307528853, "clip_ratio/high_mean": 0.007218867307528853, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/region_mean": 0.008781367330811918, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 139.25, "completions/mean_terminated_length": 86.00000762939453, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.1982464650645852, "epoch": 0.029280636221231473, "frac_reward_zero_std": 0.0, "grad_norm": 2.659188747406006, "learning_rate": 7.793939393939394e-06, "loss": -0.1369, "num_tokens": 1609182.0, "reward": 0.6998236179351807, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.700344443321228, "reward_meter_std": 0.42837557196617126, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4293443262577057, "reward_total_composite_mean": 0.6998236179351807, "reward_total_composite_std": 0.4293443560600281, "reward_total_mean": 0.6998236179351807, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.700344443321228, "rewards/meter/std": 0.42837557196617126, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6998236179351807, "rewards/total_composite/std": 0.4293443560600281, "sampling/importance_sampling_ratio/max": 1.5019832849502563, "sampling/importance_sampling_ratio/mean": 1.0047773122787476, "sampling/importance_sampling_ratio/min": 0.5312735438346863, "sampling/sampling_logp_difference/max": 0.6324782371520996, "sampling/sampling_logp_difference/mean": 0.022995855659246445, "step": 729 }, { "clip_ratio/high_max": 0.010773762594908476, "clip_ratio/high_mean": 0.010773762594908476, "clip_ratio/low_mean": 0.002890269970521331, "clip_ratio/low_min": 0.002890269970521331, "clip_ratio/region_mean": 0.013664032565429807, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 84.125, "completions/mean_terminated_length": 84.125, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.13521220535039902, "epoch": 0.029320801703016427, "frac_reward_zero_std": 0.0, "grad_norm": 3.446493625640869, "learning_rate": 7.790909090909092e-06, "loss": 0.0222, "num_tokens": 1611175.0, "reward": 0.9330159425735474, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9330159425735474, "reward_meter_std": 0.04698435589671135, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04698435962200165, "reward_total_composite_mean": 0.9330159425735474, "reward_total_composite_std": 0.04698435589671135, "reward_total_mean": 0.9330159425735474, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9330159425735474, "rewards/meter/std": 0.04698435589671135, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9330159425735474, "rewards/total_composite/std": 0.04698435589671135, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0041048526763916, "sampling/importance_sampling_ratio/min": 0.3314521312713623, "sampling/sampling_logp_difference/max": 1.4188241958618164, "sampling/sampling_logp_difference/mean": 0.02119125984609127, "step": 730 }, { "clip_ratio/high_max": 0.0005307855899445713, "clip_ratio/high_mean": 0.0005307855899445713, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0005307855899445713, "completions/clipped_ratio": 0.875, "completions/max_length": 512.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 506.875, "completions/mean_terminated_length": 471.0, "completions/min_length": 471.0, "completions/min_terminated_length": 471.0, "entropy": 0.005113576073199511, "epoch": 0.02936096718480138, "frac_reward_zero_std": 0.0, "grad_norm": 3.5135695934295654, "learning_rate": 7.787878787878789e-06, "loss": -0.2761, "num_tokens": 1613214.0, "reward": 0.055620092898607254, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5277777910232544, "reward_count_adherence_std": 0.051434461027383804, "reward_meter_mean": 0.9958064556121826, "reward_meter_std": 0.0007912082364782691, "reward_repeat_penalty_mean": 0.11126373708248138, "reward_repeat_penalty_std": 0.07416856288909912, "reward_std": 0.02950124815106392, "reward_total_composite_mean": 0.055620092898607254, "reward_total_composite_std": 0.029501251876354218, "reward_total_mean": 0.055620092898607254, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5277777910232544, "rewards/count_adherence/std": 0.051434461027383804, "rewards/meter/mean": 0.9958064556121826, "rewards/meter/std": 0.0007912082364782691, "rewards/repeat_penalty/mean": 0.11126373708248138, "rewards/repeat_penalty/std": 0.07416856288909912, "rewards/total_composite/mean": 0.055620092898607254, "rewards/total_composite/std": 0.029501251876354218, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9985577464103699, "sampling/importance_sampling_ratio/min": 0.2603153586387634, "sampling/sampling_logp_difference/max": 1.3458614349365234, "sampling/sampling_logp_difference/mean": 0.013655466958880424, "step": 731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.029401132666586335, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 7.784848484848484e-06, "loss": 0.0, "num_tokens": 1614982.0, "reward": 0.18774111568927765, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0345032773911953, "reward_meter_mean": 0.9962900280952454, "reward_meter_std": 0.001235451316460967, "reward_repeat_penalty_mean": 0.19743433594703674, "reward_repeat_penalty_std": 0.17679214477539062, "reward_std": 0.16288605332374573, "reward_total_composite_mean": 0.18774111568927765, "reward_total_composite_std": 0.16288606822490692, "reward_total_mean": 0.18774111568927765, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0345032773911953, "rewards/meter/mean": 0.9962900280952454, "rewards/meter/std": 0.001235451316460967, "rewards/repeat_penalty/mean": 0.19743433594703674, "rewards/repeat_penalty/std": 0.17679214477539062, "rewards/total_composite/mean": 0.18774111568927765, "rewards/total_composite/std": 0.16288606822490692, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 732 }, { "clip_ratio/high_max": 0.027756539173424244, "clip_ratio/high_mean": 0.027756539173424244, "clip_ratio/low_mean": 0.034895967692136765, "clip_ratio/low_min": 0.034895967692136765, "clip_ratio/region_mean": 0.06265250686556101, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 72.0, "completions/mean_terminated_length": 72.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.3269932046532631, "epoch": 0.02944129814837129, "frac_reward_zero_std": 0.0, "grad_norm": 5.411409378051758, "learning_rate": 7.781818181818183e-06, "loss": 0.026, "num_tokens": 1616806.0, "reward": 0.5835638046264648, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6948316097259521, "reward_meter_std": 0.2794814109802246, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_std": 0.28273844718933105, "reward_total_composite_mean": 0.5835638046264648, "reward_total_composite_std": 0.28273841738700867, "reward_total_mean": 0.5835638046264648, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6948316097259521, "rewards/meter/std": 0.2794814109802246, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.5835638046264648, "rewards/total_composite/std": 0.28273841738700867, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0083332061767578, "sampling/importance_sampling_ratio/min": 0.03945096582174301, "sampling/sampling_logp_difference/max": 3.232696771621704, "sampling/sampling_logp_difference/mean": 0.055757418274879456, "step": 733 }, { "clip_ratio/high_max": 0.0010775862028822303, "clip_ratio/high_mean": 0.0010775862028822303, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/region_mean": 0.003232758608646691, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 116.0, "completions/mean_terminated_length": 116.0, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.025226276833564043, "epoch": 0.029481463630156243, "frac_reward_zero_std": 0.0, "grad_norm": 1.1426414251327515, "learning_rate": 7.778787878787879e-06, "loss": 0.0012, "num_tokens": 1619174.0, "reward": 0.8406277894973755, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9807324409484863, "reward_meter_std": 0.005540105979889631, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004748655948787928, "reward_total_composite_mean": 0.8406277894973755, "reward_total_composite_std": 0.004748670384287834, "reward_total_mean": 0.8406277894973755, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9807324409484863, "rewards/meter/std": 0.005540105979889631, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8406277894973755, "rewards/total_composite/std": 0.004748670384287834, "sampling/importance_sampling_ratio/max": 1.4202854633331299, "sampling/importance_sampling_ratio/mean": 1.0012872219085693, "sampling/importance_sampling_ratio/min": 0.4178299903869629, "sampling/sampling_logp_difference/max": 0.8726806640625, "sampling/sampling_logp_difference/mean": 0.004732904955744743, "step": 734 }, { "clip_ratio/high_max": 0.011470985249616206, "clip_ratio/high_mean": 0.011470985249616206, "clip_ratio/low_mean": 0.026082158088684082, "clip_ratio/low_min": 0.026082158088684082, "clip_ratio/region_mean": 0.03755314333830029, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 77.5, "completions/mean_terminated_length": 77.5, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2497086301445961, "epoch": 0.029521629111941197, "frac_reward_zero_std": 0.0, "grad_norm": 6.164267539978027, "learning_rate": 7.775757575757576e-06, "loss": 0.0021, "num_tokens": 1620938.0, "reward": 0.7734056115150452, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8432132005691528, "reward_meter_std": 0.11800947040319443, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.09497448056936264, "reward_total_composite_mean": 0.7734056115150452, "reward_total_composite_std": 0.09497449547052383, "reward_total_mean": 0.7734056115150452, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8432132005691528, "rewards/meter/std": 0.11800947040319443, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7734056115150452, "rewards/total_composite/std": 0.09497449547052383, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0057860612869263, "sampling/importance_sampling_ratio/min": 0.2621157169342041, "sampling/sampling_logp_difference/max": 1.3389692306518555, "sampling/sampling_logp_difference/mean": 0.04740333557128906, "step": 735 }, { "clip_ratio/high_max": 0.01677176496013999, "clip_ratio/high_mean": 0.01677176496013999, "clip_ratio/low_mean": 0.008429729146882892, "clip_ratio/low_min": 0.008429729146882892, "clip_ratio/region_mean": 0.02520149410702288, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 115.875, "completions/mean_terminated_length": 115.875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 1.2863622568547726, "epoch": 0.02956179459372615, "frac_reward_zero_std": 0.0, "grad_norm": 5.787602424621582, "learning_rate": 7.772727272727273e-06, "loss": 0.46, "num_tokens": 1623113.0, "reward": 0.6369646191596985, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.7237052917480469, "reward_meter_std": 0.3117098808288574, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.40405309200286865, "reward_total_composite_mean": 0.6369646191596985, "reward_total_composite_std": 0.40405309200286865, "reward_total_mean": 0.6369646191596985, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.7237052917480469, "rewards/meter/std": 0.3117098808288574, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6369646191596985, "rewards/total_composite/std": 0.40405309200286865, "sampling/importance_sampling_ratio/max": 1.964148998260498, "sampling/importance_sampling_ratio/mean": 1.0180284976959229, "sampling/importance_sampling_ratio/min": 0.2670920789241791, "sampling/sampling_logp_difference/max": 1.3201618194580078, "sampling/sampling_logp_difference/mean": 0.08920754492282867, "step": 736 }, { "clip_ratio/high_max": 0.015350477071478963, "clip_ratio/high_mean": 0.015350477071478963, "clip_ratio/low_mean": 0.0008333333535119891, "clip_ratio/low_min": 0.0008333333535119891, "clip_ratio/region_mean": 0.016183810424990952, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 191.0, "completions/mean_length": 294.875, "completions/mean_terminated_length": 164.60000610351562, "completions/min_length": 147.0, "completions/min_terminated_length": 147.0, "entropy": 0.12422919739037752, "epoch": 0.029601960075511104, "frac_reward_zero_std": 0.0, "grad_norm": 1.1047923564910889, "learning_rate": 7.76969696969697e-06, "loss": -0.2132, "num_tokens": 1625144.0, "reward": 0.20652014017105103, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.22160132229328156, "reward_meter_mean": 0.3803099989891052, "reward_meter_std": 0.2973701059818268, "reward_repeat_penalty_mean": 0.797619104385376, "reward_repeat_penalty_std": 0.2185886949300766, "reward_std": 0.19584397971630096, "reward_total_composite_mean": 0.20652014017105103, "reward_total_composite_std": 0.19584397971630096, "reward_total_mean": 0.20652014017105103, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.22160132229328156, "rewards/meter/mean": 0.3803099989891052, "rewards/meter/std": 0.2973701059818268, "rewards/repeat_penalty/mean": 0.797619104385376, "rewards/repeat_penalty/std": 0.2185886949300766, "rewards/total_composite/mean": 0.20652014017105103, "rewards/total_composite/std": 0.19584397971630096, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015567541122437, "sampling/importance_sampling_ratio/min": 0.2301432192325592, "sampling/sampling_logp_difference/max": 1.4690535068511963, "sampling/sampling_logp_difference/mean": 0.030263110995292664, "step": 737 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.02964212555729606, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 7.766666666666666e-06, "loss": 0.0, "num_tokens": 1626768.0, "reward": 0.2381971776485443, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.050507619976997375, "reward_meter_mean": 0.9730854630470276, "reward_meter_std": 0.03588658943772316, "reward_repeat_penalty_mean": 0.2781907618045807, "reward_repeat_penalty_std": 0.2387264519929886, "reward_std": 0.20977550745010376, "reward_total_composite_mean": 0.2381971776485443, "reward_total_composite_std": 0.20977550745010376, "reward_total_mean": 0.2381971776485443, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.050507619976997375, "rewards/meter/mean": 0.9730854630470276, "rewards/meter/std": 0.03588658943772316, "rewards/repeat_penalty/mean": 0.2781907618045807, "rewards/repeat_penalty/std": 0.2387264519929886, "rewards/total_composite/mean": 0.2381971776485443, "rewards/total_composite/std": 0.20977550745010376, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 738 }, { "clip_ratio/high_max": 0.004910050658509135, "clip_ratio/high_mean": 0.004910050658509135, "clip_ratio/low_mean": 0.005539899080758914, "clip_ratio/low_min": 0.005539899080758914, "clip_ratio/region_mean": 0.01044994973926805, "completions/clipped_ratio": 0.0, "completions/max_length": 483.0, "completions/max_terminated_length": 483.0, "completions/mean_length": 423.375, "completions/mean_terminated_length": 423.375, "completions/min_length": 398.0, "completions/min_terminated_length": 398.0, "entropy": 0.0637149391695857, "epoch": 0.029682291039081012, "frac_reward_zero_std": 0.0, "grad_norm": 1.4499318599700928, "learning_rate": 7.763636363636364e-06, "loss": 0.0363, "num_tokens": 1631595.0, "reward": 0.15555939078330994, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.3888888955116272, "reward_count_adherence_std": 0.059391383081674576, "reward_meter_mean": 0.7534134387969971, "reward_meter_std": 0.2928631901741028, "reward_repeat_penalty_mean": 0.5623973608016968, "reward_repeat_penalty_std": 0.18699996173381805, "reward_std": 0.07782954722642899, "reward_total_composite_mean": 0.15555939078330994, "reward_total_composite_std": 0.07782954722642899, "reward_total_mean": 0.15555939078330994, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.3888888955116272, "rewards/count_adherence/std": 0.059391383081674576, "rewards/meter/mean": 0.7534134387969971, "rewards/meter/std": 0.2928631901741028, "rewards/repeat_penalty/mean": 0.5623973608016968, "rewards/repeat_penalty/std": 0.18699996173381805, "rewards/total_composite/mean": 0.15555939078330994, "rewards/total_composite/std": 0.07782954722642899, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9997313022613525, "sampling/importance_sampling_ratio/min": 0.0381552055478096, "sampling/sampling_logp_difference/max": 3.2660930156707764, "sampling/sampling_logp_difference/mean": 0.013467294164001942, "step": 739 }, { "clip_ratio/high_max": 0.009530608775094151, "clip_ratio/high_mean": 0.009530608775094151, "clip_ratio/low_mean": 0.008456611772999167, "clip_ratio/low_min": 0.008456611772999167, "clip_ratio/region_mean": 0.01798722054809332, "completions/clipped_ratio": 0.0, "completions/max_length": 142.0, "completions/max_terminated_length": 142.0, "completions/mean_length": 133.625, "completions/mean_terminated_length": 133.625, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.15879025869071484, "epoch": 0.029722456520865966, "frac_reward_zero_std": 0.0, "grad_norm": 3.1338510513305664, "learning_rate": 7.76060606060606e-06, "loss": -0.0085, "num_tokens": 1634096.0, "reward": 0.0800618976354599, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.11201652884483337, "reward_meter_std": 0.19264455139636993, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.11921756714582443, "reward_std": 0.13757191598415375, "reward_total_composite_mean": 0.0800618976354599, "reward_total_composite_std": 0.13757191598415375, "reward_total_mean": 0.0800618976354599, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.11201652884483337, "rewards/meter/std": 0.19264455139636993, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.11921756714582443, "rewards/total_composite/mean": 0.0800618976354599, "rewards/total_composite/std": 0.13757191598415375, "sampling/importance_sampling_ratio/max": 1.8425383567810059, "sampling/importance_sampling_ratio/mean": 1.005319595336914, "sampling/importance_sampling_ratio/min": 0.33479708433151245, "sampling/sampling_logp_difference/max": 1.0942306518554688, "sampling/sampling_logp_difference/mean": 0.02300534024834633, "step": 740 }, { "clip_ratio/high_max": 0.007794186705723405, "clip_ratio/high_mean": 0.007794186705723405, "clip_ratio/low_mean": 0.006329114083200693, "clip_ratio/low_min": 0.006329114083200693, "clip_ratio/region_mean": 0.014123300788924098, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 79.25, "completions/mean_terminated_length": 79.25, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.07466331776231527, "epoch": 0.02976262200265092, "frac_reward_zero_std": 0.0, "grad_norm": 1.2720859050750732, "learning_rate": 7.757575757575758e-06, "loss": -0.0061, "num_tokens": 1636098.0, "reward": 0.794016420841217, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9925205707550049, "reward_meter_std": 0.0022588411811739206, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0018070697551593184, "reward_total_composite_mean": 0.794016420841217, "reward_total_composite_std": 0.001807067426852882, "reward_total_mean": 0.794016420841217, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9925205707550049, "rewards/meter/std": 0.0022588411811739206, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.794016420841217, "rewards/total_composite/std": 0.001807067426852882, "sampling/importance_sampling_ratio/max": 1.5393544435501099, "sampling/importance_sampling_ratio/mean": 1.001147985458374, "sampling/importance_sampling_ratio/min": 0.4444310963153839, "sampling/sampling_logp_difference/max": 0.8109602928161621, "sampling/sampling_logp_difference/mean": 0.011411315761506557, "step": 741 }, { "clip_ratio/high_max": 0.018717249389737844, "clip_ratio/high_mean": 0.018717249389737844, "clip_ratio/low_mean": 0.021951976465061307, "clip_ratio/low_min": 0.021951976465061307, "clip_ratio/region_mean": 0.04066922585479915, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 120.75, "completions/mean_terminated_length": 120.75, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.12627906422130764, "epoch": 0.029802787484435874, "frac_reward_zero_std": 0.0, "grad_norm": 7.301284313201904, "learning_rate": 7.754545454545455e-06, "loss": 0.012, "num_tokens": 1638424.0, "reward": 0.5217785835266113, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7745586633682251, "reward_meter_std": 0.25043216347694397, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.17806050181388855, "reward_std": 0.1874629259109497, "reward_total_composite_mean": 0.5217785835266113, "reward_total_composite_std": 0.1874629557132721, "reward_total_mean": 0.5217785835266113, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7745586633682251, "rewards/meter/std": 0.25043216347694397, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.17806050181388855, "rewards/total_composite/mean": 0.5217785835266113, "rewards/total_composite/std": 0.1874629557132721, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9962561726570129, "sampling/importance_sampling_ratio/min": 0.012216109782457352, "sampling/sampling_logp_difference/max": 4.404999732971191, "sampling/sampling_logp_difference/mean": 0.04599983990192413, "step": 742 }, { "clip_ratio/high_max": 0.0012283907853998244, "clip_ratio/high_mean": 0.0012283907853998244, "clip_ratio/low_mean": 0.0017248676158487797, "clip_ratio/low_min": 0.0017248676158487797, "clip_ratio/region_mean": 0.002953258401248604, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 510.0, "completions/mean_terminated_length": 508.0, "completions/min_length": 507.0, "completions/min_terminated_length": 507.0, "entropy": 0.03308352828025818, "epoch": 0.02984295296622083, "frac_reward_zero_std": 0.0, "grad_norm": 1.0547409057617188, "learning_rate": 7.751515151515153e-06, "loss": 0.1769, "num_tokens": 1642208.0, "reward": 0.13946115970611572, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.6153846383094788, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.45572322607040405, "reward_meter_std": 0.1519794911146164, "reward_repeat_penalty_mean": 0.5480158925056458, "reward_repeat_penalty_std": 0.010451768524944782, "reward_std": 0.07740650326013565, "reward_total_composite_mean": 0.13946115970611572, "reward_total_composite_std": 0.07740650326013565, "reward_total_mean": 0.13946115970611572, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.6153846383094788, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.45572322607040405, "rewards/meter/std": 0.1519794911146164, "rewards/repeat_penalty/mean": 0.5480158925056458, "rewards/repeat_penalty/std": 0.010451768524944782, "rewards/total_composite/mean": 0.13946115970611572, "rewards/total_composite/std": 0.07740650326013565, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0025880336761475, "sampling/importance_sampling_ratio/min": 0.32132917642593384, "sampling/sampling_logp_difference/max": 1.135289192199707, "sampling/sampling_logp_difference/mean": 0.00907099712640047, "step": 743 }, { "clip_ratio/high_max": 0.027501578675583005, "clip_ratio/high_mean": 0.027501578675583005, "clip_ratio/low_mean": 0.02111415727995336, "clip_ratio/low_min": 0.02111415727995336, "clip_ratio/region_mean": 0.048615735955536366, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 64.5, "completions/mean_terminated_length": 64.5, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.6070369817316532, "epoch": 0.029883118448005785, "frac_reward_zero_std": 0.0, "grad_norm": 6.8046875, "learning_rate": 7.74848484848485e-06, "loss": 0.0283, "num_tokens": 1643996.0, "reward": 0.5396110415458679, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5396110415458679, "reward_meter_std": 0.20649512112140656, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.20649512112140656, "reward_total_composite_mean": 0.5396110415458679, "reward_total_composite_std": 0.20649512112140656, "reward_total_mean": 0.5396110415458679, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5396110415458679, "rewards/meter/std": 0.20649512112140656, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5396110415458679, "rewards/total_composite/std": 0.20649512112140656, "sampling/importance_sampling_ratio/max": 1.632293462753296, "sampling/importance_sampling_ratio/mean": 1.0097684860229492, "sampling/importance_sampling_ratio/min": 0.28896111249923706, "sampling/sampling_logp_difference/max": 1.2414631843566895, "sampling/sampling_logp_difference/mean": 0.06342907249927521, "step": 744 }, { "clip_ratio/high_max": 0.043645198456943035, "clip_ratio/high_mean": 0.043645198456943035, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/region_mean": 0.051709714345633984, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.25149067025631666, "epoch": 0.02992328392979074, "frac_reward_zero_std": 0.0, "grad_norm": 9.705404281616211, "learning_rate": 7.745454545454545e-06, "loss": 0.0099, "num_tokens": 1645756.0, "reward": 0.9103853106498718, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9103853106498718, "reward_meter_std": 0.22165298461914062, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22165298461914062, "reward_total_composite_mean": 0.9103853106498718, "reward_total_composite_std": 0.22165298461914062, "reward_total_mean": 0.9103853106498718, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9103853106498718, "rewards/meter/std": 0.22165298461914062, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9103853106498718, "rewards/total_composite/std": 0.22165298461914062, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0033738613128662, "sampling/importance_sampling_ratio/min": 0.13348090648651123, "sampling/sampling_logp_difference/max": 2.013796806335449, "sampling/sampling_logp_difference/mean": 0.07189808040857315, "step": 745 }, { "clip_ratio/high_max": 0.006430621142499149, "clip_ratio/high_mean": 0.006430621142499149, "clip_ratio/low_mean": 0.003769519622437656, "clip_ratio/low_min": 0.003769519622437656, "clip_ratio/region_mean": 0.010200140764936805, "completions/clipped_ratio": 0.0, "completions/max_length": 260.0, "completions/max_terminated_length": 260.0, "completions/mean_length": 238.25, "completions/mean_terminated_length": 238.25, "completions/min_length": 205.0, "completions/min_terminated_length": 205.0, "entropy": 0.06980151077732444, "epoch": 0.029963449411575693, "frac_reward_zero_std": 0.0, "grad_norm": 2.2934165000915527, "learning_rate": 7.742424242424244e-06, "loss": 0.0227, "num_tokens": 1649182.0, "reward": 0.2228499948978424, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.8676279783248901, "reward_meter_std": 0.3215867578983307, "reward_repeat_penalty_mean": 0.36800700426101685, "reward_repeat_penalty_std": 0.24577626585960388, "reward_std": 0.18318407237529755, "reward_total_composite_mean": 0.2228499948978424, "reward_total_composite_std": 0.18318407237529755, "reward_total_mean": 0.2228499948978424, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.8676279783248901, "rewards/meter/std": 0.3215867578983307, "rewards/repeat_penalty/mean": 0.36800700426101685, "rewards/repeat_penalty/std": 0.24577626585960388, "rewards/total_composite/mean": 0.2228499948978424, "rewards/total_composite/std": 0.18318407237529755, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0014268159866333, "sampling/importance_sampling_ratio/min": 0.016722269356250763, "sampling/sampling_logp_difference/max": 4.0910139083862305, "sampling/sampling_logp_difference/mean": 0.018860070034861565, "step": 746 }, { "clip_ratio/high_max": 0.018227352295070887, "clip_ratio/high_mean": 0.018227352295070887, "clip_ratio/low_mean": 0.012351190904155374, "clip_ratio/low_min": 0.012351190904155374, "clip_ratio/region_mean": 0.03057854319922626, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 40.625, "completions/mean_terminated_length": 40.625, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.28702736645936966, "epoch": 0.030003614893360647, "frac_reward_zero_std": 0.0, "grad_norm": 7.876811981201172, "learning_rate": 7.73939393939394e-06, "loss": -0.0252, "num_tokens": 1650683.0, "reward": 0.9663740396499634, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9663740396499634, "reward_meter_std": 0.011263493448495865, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011263499036431313, "reward_total_composite_mean": 0.9663740396499634, "reward_total_composite_std": 0.011263493448495865, "reward_total_mean": 0.9663740396499634, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9663740396499634, "rewards/meter/std": 0.011263493448495865, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9663740396499634, "rewards/total_composite/std": 0.011263493448495865, "sampling/importance_sampling_ratio/max": 1.344473123550415, "sampling/importance_sampling_ratio/mean": 1.0029720067977905, "sampling/importance_sampling_ratio/min": 0.41253525018692017, "sampling/sampling_logp_difference/max": 0.8854336738586426, "sampling/sampling_logp_difference/mean": 0.03788968175649643, "step": 747 }, { "clip_ratio/high_max": 0.021474606008268893, "clip_ratio/high_mean": 0.021474606008268893, "clip_ratio/low_mean": 0.002961171790957451, "clip_ratio/low_min": 0.002961171790957451, "clip_ratio/region_mean": 0.024435777799226344, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 82.125, "completions/mean_terminated_length": 82.125, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.22462552785873413, "epoch": 0.0300437803751456, "frac_reward_zero_std": 0.0, "grad_norm": 4.674757957458496, "learning_rate": 7.736363636363637e-06, "loss": 0.031, "num_tokens": 1652564.0, "reward": 0.8491703271865845, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8491703271865845, "reward_meter_std": 0.15907908976078033, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.15907907485961914, "reward_total_composite_mean": 0.8491703271865845, "reward_total_composite_std": 0.15907908976078033, "reward_total_mean": 0.8491703271865845, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8491703271865845, "rewards/meter/std": 0.15907908976078033, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8491703271865845, "rewards/total_composite/std": 0.15907908976078033, "sampling/importance_sampling_ratio/max": 1.843922734260559, "sampling/importance_sampling_ratio/mean": 1.0036686658859253, "sampling/importance_sampling_ratio/min": 0.2724129855632782, "sampling/sampling_logp_difference/max": 1.300436019897461, "sampling/sampling_logp_difference/mean": 0.0295439250767231, "step": 748 }, { "clip_ratio/high_max": 0.008620689623057842, "clip_ratio/high_mean": 0.008620689623057842, "clip_ratio/low_mean": 0.026110241888090968, "clip_ratio/low_min": 0.026110241888090968, "clip_ratio/region_mean": 0.03473093151114881, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 93.625, "completions/mean_terminated_length": 93.625, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.12411046819761395, "epoch": 0.030083945856930555, "frac_reward_zero_std": 0.0, "grad_norm": 6.799373626708984, "learning_rate": 7.733333333333334e-06, "loss": 0.0372, "num_tokens": 1654625.0, "reward": 0.8075714707374573, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9564855694770813, "reward_meter_std": 0.10235674679279327, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.0810769572854042, "reward_total_composite_mean": 0.8075714707374573, "reward_total_composite_std": 0.0810769721865654, "reward_total_mean": 0.8075714707374573, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9564855694770813, "rewards/meter/std": 0.10235674679279327, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8075714707374573, "rewards/total_composite/std": 0.0810769721865654, "sampling/importance_sampling_ratio/max": 1.903843641281128, "sampling/importance_sampling_ratio/mean": 0.9965777397155762, "sampling/importance_sampling_ratio/min": 0.08325432986021042, "sampling/sampling_logp_difference/max": 2.4858551025390625, "sampling/sampling_logp_difference/mean": 0.03427453711628914, "step": 749 }, { "clip_ratio/high_max": 0.022817256744019687, "clip_ratio/high_mean": 0.022817256744019687, "clip_ratio/low_mean": 0.00892857147846371, "clip_ratio/low_min": 0.00892857147846371, "clip_ratio/region_mean": 0.031745828222483397, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.21013184823095798, "epoch": 0.03012411133871551, "frac_reward_zero_std": 0.0, "grad_norm": 6.688387870788574, "learning_rate": 7.730303030303032e-06, "loss": 0.031, "num_tokens": 1656513.0, "reward": 0.8990964889526367, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9406866431236267, "reward_meter_std": 0.1574798822402954, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.18213684856891632, "reward_total_composite_mean": 0.8990964889526367, "reward_total_composite_std": 0.18213681876659393, "reward_total_mean": 0.8990964889526367, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9406866431236267, "rewards/meter/std": 0.1574798822402954, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8990964889526367, "rewards/total_composite/std": 0.18213681876659393, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0121128559112549, "sampling/importance_sampling_ratio/min": 0.42169255018234253, "sampling/sampling_logp_difference/max": 0.9646925926208496, "sampling/sampling_logp_difference/mean": 0.035663776099681854, "step": 750 }, { "epoch": 0.03012411133871551, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.23076923076923078, "eval_completions/max_length": 496.7692307692308, "eval_completions/max_terminated_length": 393.46153846153845, "eval_completions/mean_length": 284.52884615384613, "eval_completions/mean_terminated_length": 215.78288092980017, "eval_completions/min_length": 61.46153846153846, "eval_completions/min_terminated_length": 61.46153846153846, "eval_entropy": 0.17351046003974402, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 1656513.0, "eval_reward": 0.33899185634576356, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_count_adherence_mean": 0.7424245018225449, "eval_reward_count_adherence_std": 0.2582179422561939, "eval_reward_meter_mean": 0.6660121427132533, "eval_reward_meter_std": 0.37975076070198643, "eval_reward_repeat_penalty_mean": 0.638540084545429, "eval_reward_repeat_penalty_std": 0.270676647241299, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.33899185634576356, "eval_reward_total_composite_std": 0.3308297275350644, "eval_reward_total_mean": 0.33899185634576356, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/count_adherence/mean": 0.7424245018225449, "eval_rewards/count_adherence/std": 0.2582179422561939, "eval_rewards/meter/mean": 0.6660121427132533, "eval_rewards/meter/std": 0.37975076070198643, "eval_rewards/repeat_penalty/mean": 0.638540084545429, "eval_rewards/repeat_penalty/std": 0.270676647241299, "eval_rewards/total_composite/mean": 0.33899185634576356, "eval_rewards/total_composite/std": 0.3308297275350644, "eval_runtime": 93.1633, "eval_samples_per_second": 1.116, "eval_sampling/importance_sampling_ratio/max": 1.3942875678722675, "eval_sampling/importance_sampling_ratio/mean": 1.0036690326837392, "eval_sampling/importance_sampling_ratio/min": 0.4615517258644104, "eval_sampling/sampling_logp_difference/max": 0.8113565261547382, "eval_sampling/sampling_logp_difference/mean": 0.012646582407447008, "eval_steps_per_second": 0.14, "step": 750 }, { "clip_ratio/high_max": 0.008403955027461052, "clip_ratio/high_mean": 0.008403955027461052, "clip_ratio/low_mean": 0.006320621585473418, "clip_ratio/low_min": 0.006320621585473418, "clip_ratio/region_mean": 0.01472457661293447, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 59.25, "completions/mean_terminated_length": 59.25, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.052004152443259954, "epoch": 0.030164276820500463, "frac_reward_zero_std": 0.0, "grad_norm": 5.31432580947876, "learning_rate": 7.727272727272727e-06, "loss": 0.0024, "num_tokens": 1658395.0, "reward": 0.9893741607666016, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9893741607666016, "reward_meter_std": 0.007569636683911085, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007569642271846533, "reward_total_composite_mean": 0.9893741607666016, "reward_total_composite_std": 0.007569636683911085, "reward_total_mean": 0.9893741607666016, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9893741607666016, "rewards/meter/std": 0.007569636683911085, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9893741607666016, "rewards/total_composite/std": 0.007569636683911085, "sampling/importance_sampling_ratio/max": 1.868045449256897, "sampling/importance_sampling_ratio/mean": 1.0002750158309937, "sampling/importance_sampling_ratio/min": 0.3949167728424072, "sampling/sampling_logp_difference/max": 0.9290802478790283, "sampling/sampling_logp_difference/mean": 0.013296845369040966, "step": 751 }, { "clip_ratio/high_max": 0.006320621585473418, "clip_ratio/high_mean": 0.006320621585473418, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/region_mean": 0.008403955027461052, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 59.25, "completions/mean_terminated_length": 59.25, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.03427604655735195, "epoch": 0.030204442302285417, "frac_reward_zero_std": 0.0, "grad_norm": 4.07772970199585, "learning_rate": 7.724242424242424e-06, "loss": 0.0051, "num_tokens": 1660117.0, "reward": 0.9534660577774048, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949396848678589, "reward_meter_std": 0.00028474157443270087, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11713218688964844, "reward_total_composite_mean": 0.9534660577774048, "reward_total_composite_std": 0.11713220924139023, "reward_total_mean": 0.9534660577774048, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949396848678589, "rewards/meter/std": 0.00028474157443270087, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9534660577774048, "rewards/total_composite/std": 0.11713220924139023, "sampling/importance_sampling_ratio/max": 1.3358639478683472, "sampling/importance_sampling_ratio/mean": 1.0019704103469849, "sampling/importance_sampling_ratio/min": 0.2900037169456482, "sampling/sampling_logp_difference/max": 1.2378616333007812, "sampling/sampling_logp_difference/mean": 0.009837880730628967, "step": 752 }, { "clip_ratio/high_max": 0.04496192745864391, "clip_ratio/high_mean": 0.04496192745864391, "clip_ratio/low_mean": 0.017439668532460928, "clip_ratio/low_min": 0.017439668532460928, "clip_ratio/region_mean": 0.06240159599110484, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 35.875, "completions/mean_terminated_length": 35.875, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.6193088293075562, "epoch": 0.03024460778407037, "frac_reward_zero_std": 0.0, "grad_norm": 14.107855796813965, "learning_rate": 7.721212121212122e-06, "loss": -0.0196, "num_tokens": 1661428.0, "reward": 0.9349486827850342, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9349486827850342, "reward_meter_std": 0.1060987114906311, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1060987114906311, "reward_total_composite_mean": 0.9349486827850342, "reward_total_composite_std": 0.1060987114906311, "reward_total_mean": 0.9349486827850342, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9349486827850342, "rewards/meter/std": 0.1060987114906311, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9349486827850342, "rewards/total_composite/std": 0.1060987114906311, "sampling/importance_sampling_ratio/max": 1.6115726232528687, "sampling/importance_sampling_ratio/mean": 1.0009469985961914, "sampling/importance_sampling_ratio/min": 0.20662181079387665, "sampling/sampling_logp_difference/max": 1.5768651962280273, "sampling/sampling_logp_difference/mean": 0.07551710307598114, "step": 753 }, { "clip_ratio/high_max": 0.018074912950396538, "clip_ratio/high_mean": 0.018074912950396538, "clip_ratio/low_mean": 0.008928571594879031, "clip_ratio/low_min": 0.008928571594879031, "clip_ratio/region_mean": 0.02700348454527557, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 41.5, "completions/mean_terminated_length": 41.5, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.11915541160851717, "epoch": 0.030284773265855325, "frac_reward_zero_std": 0.0, "grad_norm": 2.418649673461914, "learning_rate": 7.718181818181819e-06, "loss": 0.0091, "num_tokens": 1663080.0, "reward": 0.9943797588348389, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943797588348389, "reward_meter_std": 0.0011224248446524143, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0011224271729588509, "reward_total_composite_mean": 0.9943797588348389, "reward_total_composite_std": 0.0011224248446524143, "reward_total_mean": 0.9943797588348389, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943797588348389, "rewards/meter/std": 0.0011224248446524143, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943797588348389, "rewards/total_composite/std": 0.0011224248446524143, "sampling/importance_sampling_ratio/max": 1.2741270065307617, "sampling/importance_sampling_ratio/mean": 0.9925946593284607, "sampling/importance_sampling_ratio/min": 0.2249595671892166, "sampling/sampling_logp_difference/max": 1.4918346405029297, "sampling/sampling_logp_difference/mean": 0.030370449647307396, "step": 754 }, { "clip_ratio/high_max": 0.012195121496915817, "clip_ratio/high_mean": 0.012195121496915817, "clip_ratio/low_mean": 0.0030487803742289543, "clip_ratio/low_min": 0.0030487803742289543, "clip_ratio/region_mean": 0.015243901871144772, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 41.0, "completions/mean_terminated_length": 41.0, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.10165042616426945, "epoch": 0.03032493874764028, "frac_reward_zero_std": 0.0, "grad_norm": 5.144253730773926, "learning_rate": 7.715151515151516e-06, "loss": -0.0018, "num_tokens": 1664616.0, "reward": 0.9944278001785278, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9944278001785278, "reward_meter_std": 0.0012945194030180573, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0012945224298164248, "reward_total_composite_mean": 0.9944278001785278, "reward_total_composite_std": 0.0012945194030180573, "reward_total_mean": 0.9944278001785278, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9944278001785278, "rewards/meter/std": 0.0012945194030180573, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944278001785278, "rewards/total_composite/std": 0.0012945194030180573, "sampling/importance_sampling_ratio/max": 1.2102103233337402, "sampling/importance_sampling_ratio/mean": 1.0021907091140747, "sampling/importance_sampling_ratio/min": 0.5395178198814392, "sampling/sampling_logp_difference/max": 0.617079496383667, "sampling/sampling_logp_difference/mean": 0.014835118316113949, "step": 755 }, { "clip_ratio/high_max": 0.005401917500421405, "clip_ratio/high_mean": 0.005401917500421405, "clip_ratio/low_mean": 0.005512091098353267, "clip_ratio/low_min": 0.005512091098353267, "clip_ratio/region_mean": 0.010914008598774672, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 113.75, "completions/mean_terminated_length": 113.75, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.0643842932768166, "epoch": 0.030365104229425233, "frac_reward_zero_std": 0.0, "grad_norm": 2.4242031574249268, "learning_rate": 7.712121212121213e-06, "loss": -0.0099, "num_tokens": 1666878.0, "reward": 0.6285786032676697, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9662827253341675, "reward_meter_std": 0.009305375628173351, "reward_repeat_penalty_mean": 0.6500000357627869, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09414192289113998, "reward_total_composite_mean": 0.6285786032676697, "reward_total_composite_std": 0.0941418930888176, "reward_total_mean": 0.6285786032676697, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9662827253341675, "rewards/meter/std": 0.009305375628173351, "rewards/repeat_penalty/mean": 0.6500000357627869, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6285786032676697, "rewards/total_composite/std": 0.0941418930888176, "sampling/importance_sampling_ratio/max": 1.911029577255249, "sampling/importance_sampling_ratio/mean": 1.0027672052383423, "sampling/importance_sampling_ratio/min": 0.20104746520519257, "sampling/sampling_logp_difference/max": 1.6042141914367676, "sampling/sampling_logp_difference/mean": 0.011911381967365742, "step": 756 }, { "clip_ratio/high_max": 0.007282168400706723, "clip_ratio/high_mean": 0.007282168400706723, "clip_ratio/low_mean": 0.0010245901066809893, "clip_ratio/low_min": 0.0010245901066809893, "clip_ratio/region_mean": 0.008306758507387713, "completions/clipped_ratio": 0.0, "completions/max_length": 267.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 243.25, "completions/mean_terminated_length": 243.25, "completions/min_length": 217.0, "completions/min_terminated_length": 217.0, "entropy": 0.06725997012108564, "epoch": 0.030405269711210187, "frac_reward_zero_std": 0.0, "grad_norm": 1.2799259424209595, "learning_rate": 7.709090909090909e-06, "loss": 0.0259, "num_tokens": 1670368.0, "reward": 0.2822417914867401, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6500000357627869, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.9969884157180786, "reward_meter_std": 0.000910833477973938, "reward_repeat_penalty_mean": 0.4187062978744507, "reward_repeat_penalty_std": 0.23771634697914124, "reward_std": 0.18207496404647827, "reward_total_composite_mean": 0.2822417914867401, "reward_total_composite_std": 0.18207497894763947, "reward_total_mean": 0.2822417914867401, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6500000357627869, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.9969884157180786, "rewards/meter/std": 0.000910833477973938, "rewards/repeat_penalty/mean": 0.4187062978744507, "rewards/repeat_penalty/std": 0.23771634697914124, "rewards/total_composite/mean": 0.2822417914867401, "rewards/total_composite/std": 0.18207497894763947, "sampling/importance_sampling_ratio/max": 1.5265671014785767, "sampling/importance_sampling_ratio/mean": 1.0003676414489746, "sampling/importance_sampling_ratio/min": 0.27827587723731995, "sampling/sampling_logp_difference/max": 1.2791423797607422, "sampling/sampling_logp_difference/mean": 0.00974962953478098, "step": 757 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.03044543519299514, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 7.706060606060606e-06, "loss": 0.0, "num_tokens": 1672112.0, "reward": 0.059182293713092804, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.15625, "reward_count_adherence_std": 0.11080066114664078, "reward_meter_mean": 0.8747541308403015, "reward_meter_std": 0.1777074635028839, "reward_repeat_penalty_mean": 0.42480412125587463, "reward_repeat_penalty_std": 0.2929826080799103, "reward_std": 0.06593845039606094, "reward_total_composite_mean": 0.059182293713092804, "reward_total_composite_std": 0.06593845039606094, "reward_total_mean": 0.059182293713092804, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.15625, "rewards/count_adherence/std": 0.11080066114664078, "rewards/meter/mean": 0.8747541308403015, "rewards/meter/std": 0.1777074635028839, "rewards/repeat_penalty/mean": 0.42480412125587463, "rewards/repeat_penalty/std": 0.2929826080799103, "rewards/total_composite/mean": 0.059182293713092804, "rewards/total_composite/std": 0.06593845039606094, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 758 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 87.0, "completions/mean_terminated_length": 87.0, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.014668526826426387, "epoch": 0.030485600674780094, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.703030303030304e-06, "loss": 0.0, "num_tokens": 1674144.0, "reward": 0.5969825983047485, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949710369110107, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.5969825983047485, "reward_total_composite_std": 0.0, "reward_total_mean": 0.5969825983047485, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949710369110107, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5969825983047485, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0613410472869873, "sampling/importance_sampling_ratio/mean": 1.0006695985794067, "sampling/importance_sampling_ratio/min": 0.848580002784729, "sampling/sampling_logp_difference/max": 0.1641908884048462, "sampling/sampling_logp_difference/mean": 0.001654994674026966, "step": 759 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/region_mean": 0.004629629664123058, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 54.125, "completions/mean_terminated_length": 54.125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.042751661501824856, "epoch": 0.03052576615656505, "frac_reward_zero_std": 0.0, "grad_norm": 4.493424415588379, "learning_rate": 7.7e-06, "loss": -0.0037, "num_tokens": 1675945.0, "reward": 0.6740554571151733, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9517734050750732, "reward_meter_std": 0.0011382178636267781, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_std": 0.11107677221298218, "reward_total_composite_mean": 0.6740554571151733, "reward_total_composite_std": 0.11107677966356277, "reward_total_mean": 0.6740554571151733, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9517734050750732, "rewards/meter/std": 0.0011382178636267781, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.6740554571151733, "rewards/total_composite/std": 0.11107677966356277, "sampling/importance_sampling_ratio/max": 1.612492322921753, "sampling/importance_sampling_ratio/mean": 1.0033941268920898, "sampling/importance_sampling_ratio/min": 0.7561957836151123, "sampling/sampling_logp_difference/max": 0.4777810573577881, "sampling/sampling_logp_difference/mean": 0.005725136026740074, "step": 760 }, { "clip_ratio/high_max": 0.0021186440717428923, "clip_ratio/high_mean": 0.0021186440717428923, "clip_ratio/low_mean": 0.006428988883271813, "clip_ratio/low_min": 0.006428988883271813, "clip_ratio/region_mean": 0.008547632955014706, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 58.25, "completions/mean_terminated_length": 58.25, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.05586709058843553, "epoch": 0.030565931638350002, "frac_reward_zero_std": 0.0, "grad_norm": 0.8903390765190125, "learning_rate": 7.696969696969696e-06, "loss": -0.0023, "num_tokens": 1677747.0, "reward": 0.7047345042228699, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949068427085876, "reward_meter_std": 0.0007968654972501099, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_std": 0.1173337996006012, "reward_total_composite_mean": 0.7047345042228699, "reward_total_composite_std": 0.1173337996006012, "reward_total_mean": 0.7047345042228699, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949068427085876, "rewards/meter/std": 0.0007968654972501099, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.7047345042228699, "rewards/total_composite/std": 0.1173337996006012, "sampling/importance_sampling_ratio/max": 1.6007080078125, "sampling/importance_sampling_ratio/mean": 1.0014004707336426, "sampling/importance_sampling_ratio/min": 0.3485388159751892, "sampling/sampling_logp_difference/max": 1.0540056228637695, "sampling/sampling_logp_difference/mean": 0.011640225537121296, "step": 761 }, { "clip_ratio/high_max": 0.006666666595265269, "clip_ratio/high_mean": 0.006666666595265269, "clip_ratio/low_mean": 0.012610982405021787, "clip_ratio/low_min": 0.012610982405021787, "clip_ratio/region_mean": 0.019277649000287056, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 78.25, "completions/mean_terminated_length": 78.25, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.11404913989827037, "epoch": 0.030606097120134956, "frac_reward_zero_std": 0.0, "grad_norm": 5.6463236808776855, "learning_rate": 7.693939393939395e-06, "loss": 0.024, "num_tokens": 1679613.0, "reward": 0.6016985774040222, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8964411020278931, "reward_meter_std": 0.12688924372196198, "reward_repeat_penalty_mean": 0.6750000715255737, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.11093335598707199, "reward_total_composite_mean": 0.6016985774040222, "reward_total_composite_std": 0.1109333410859108, "reward_total_mean": 0.6016985774040222, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8964411020278931, "rewards/meter/std": 0.12688924372196198, "rewards/repeat_penalty/mean": 0.6750000715255737, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6016985774040222, "rewards/total_composite/std": 0.1109333410859108, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0003811120986938, "sampling/importance_sampling_ratio/min": 0.4048405587673187, "sampling/sampling_logp_difference/max": 0.9042620658874512, "sampling/sampling_logp_difference/mean": 0.02804330736398697, "step": 762 }, { "clip_ratio/high_max": 0.005744358117226511, "clip_ratio/high_mean": 0.005744358117226511, "clip_ratio/low_mean": 0.005670643062330782, "clip_ratio/low_min": 0.005670643062330782, "clip_ratio/region_mean": 0.011415001179557294, "completions/clipped_ratio": 0.0, "completions/max_length": 498.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 438.125, "completions/mean_terminated_length": 438.125, "completions/min_length": 391.0, "completions/min_terminated_length": 391.0, "entropy": 0.07848789915442467, "epoch": 0.03064626260191991, "frac_reward_zero_std": 0.0, "grad_norm": 1.2447540760040283, "learning_rate": 7.690909090909091e-06, "loss": 0.0481, "num_tokens": 1684742.0, "reward": 0.12146620452404022, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.2142857164144516, "reward_count_adherence_std": 0.07636035978794098, "reward_meter_mean": 0.9878873825073242, "reward_meter_std": 0.019713442772626877, "reward_repeat_penalty_mean": 0.5576170682907104, "reward_repeat_penalty_std": 0.0946657732129097, "reward_std": 0.05504889413714409, "reward_total_composite_mean": 0.12146620452404022, "reward_total_composite_std": 0.05504889413714409, "reward_total_mean": 0.12146620452404022, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.2142857164144516, "rewards/count_adherence/std": 0.07636035978794098, "rewards/meter/mean": 0.9878873825073242, "rewards/meter/std": 0.019713442772626877, "rewards/repeat_penalty/mean": 0.5576170682907104, "rewards/repeat_penalty/std": 0.0946657732129097, "rewards/total_composite/mean": 0.12146620452404022, "rewards/total_composite/std": 0.05504889413714409, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0031057596206665, "sampling/importance_sampling_ratio/min": 0.22546334564685822, "sampling/sampling_logp_difference/max": 1.4895976781845093, "sampling/sampling_logp_difference/mean": 0.010950884781777859, "step": 763 }, { "clip_ratio/high_max": 0.02714023506268859, "clip_ratio/high_mean": 0.02714023506268859, "clip_ratio/low_mean": 0.005871212342754006, "clip_ratio/low_min": 0.005871212342754006, "clip_ratio/region_mean": 0.033011447405442595, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 63.625, "completions/mean_terminated_length": 63.625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.1180117940530181, "epoch": 0.030686428083704864, "frac_reward_zero_std": 0.0, "grad_norm": 5.640087127685547, "learning_rate": 7.687878787878788e-06, "loss": 0.0189, "num_tokens": 1686539.0, "reward": 0.7850605845451355, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.874701976776123, "reward_meter_std": 0.18508003652095795, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.2699912190437317, "reward_total_composite_mean": 0.7850605845451355, "reward_total_composite_std": 0.2699912190437317, "reward_total_mean": 0.7850605845451355, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.874701976776123, "rewards/meter/std": 0.18508003652095795, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7850605845451355, "rewards/total_composite/std": 0.2699912190437317, "sampling/importance_sampling_ratio/max": 1.438651442527771, "sampling/importance_sampling_ratio/mean": 0.9924401044845581, "sampling/importance_sampling_ratio/min": 0.16446030139923096, "sampling/sampling_logp_difference/max": 1.8050861358642578, "sampling/sampling_logp_difference/mean": 0.032904475927352905, "step": 764 }, { "clip_ratio/high_max": 0.0028409091755747795, "clip_ratio/high_mean": 0.0028409091755747795, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0028409091755747795, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 87.125, "completions/mean_terminated_length": 87.125, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.008648848335724324, "epoch": 0.030726593565489818, "frac_reward_zero_std": 0.0, "grad_norm": 4.968373775482178, "learning_rate": 7.684848484848485e-06, "loss": 0.0008, "num_tokens": 1688548.0, "reward": 0.6467622518539429, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9950021505355835, "reward_meter_std": 8.806584810372442e-05, "reward_repeat_penalty_mean": 0.6500000357627869, "reward_repeat_penalty_std": 0.1414213478565216, "reward_std": 0.14079821109771729, "reward_total_composite_mean": 0.6467622518539429, "reward_total_composite_std": 0.14079822599887848, "reward_total_mean": 0.6467622518539429, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9950021505355835, "rewards/meter/std": 8.806584810372442e-05, "rewards/repeat_penalty/mean": 0.6500000357627869, "rewards/repeat_penalty/std": 0.1414213478565216, "rewards/total_composite/mean": 0.6467622518539429, "rewards/total_composite/std": 0.14079822599887848, "sampling/importance_sampling_ratio/max": 1.767101764678955, "sampling/importance_sampling_ratio/mean": 0.9999979138374329, "sampling/importance_sampling_ratio/min": 0.1755325347185135, "sampling/sampling_logp_difference/max": 1.7399308681488037, "sampling/sampling_logp_difference/mean": 0.004592791199684143, "step": 765 }, { "clip_ratio/high_max": 0.03126057400368154, "clip_ratio/high_mean": 0.03126057400368154, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/region_mean": 0.03606826649047434, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 52.0, "completions/mean_terminated_length": 52.0, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.19545143470168114, "epoch": 0.030766759047274772, "frac_reward_zero_std": 0.0, "grad_norm": 10.79392147064209, "learning_rate": 7.681818181818183e-06, "loss": 0.0014, "num_tokens": 1690236.0, "reward": 0.8247180581092834, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8247180581092834, "reward_meter_std": 0.3005145490169525, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3005145490169525, "reward_total_composite_mean": 0.8247180581092834, "reward_total_composite_std": 0.3005145490169525, "reward_total_mean": 0.8247180581092834, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8247180581092834, "rewards/meter/std": 0.3005145490169525, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8247180581092834, "rewards/total_composite/std": 0.3005145490169525, "sampling/importance_sampling_ratio/max": 1.6643987894058228, "sampling/importance_sampling_ratio/mean": 0.997559666633606, "sampling/importance_sampling_ratio/min": 0.2000909447669983, "sampling/sampling_logp_difference/max": 1.6089832782745361, "sampling/sampling_logp_difference/mean": 0.04232970252633095, "step": 766 }, { "clip_ratio/high_max": 0.026226354064419866, "clip_ratio/high_mean": 0.026226354064419866, "clip_ratio/low_mean": 0.03220835281535983, "clip_ratio/low_min": 0.03220835281535983, "clip_ratio/region_mean": 0.058434706879779696, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 72.625, "completions/mean_terminated_length": 72.625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.46072521805763245, "epoch": 0.030806924529059726, "frac_reward_zero_std": 0.0, "grad_norm": 12.172296524047852, "learning_rate": 7.678787878787878e-06, "loss": 0.0103, "num_tokens": 1692065.0, "reward": 0.36088046431541443, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3632194995880127, "reward_meter_std": 0.32341283559799194, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3263150453567505, "reward_total_composite_mean": 0.36088046431541443, "reward_total_composite_std": 0.3263150453567505, "reward_total_mean": 0.36088046431541443, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3632194995880127, "rewards/meter/std": 0.32341283559799194, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.36088046431541443, "rewards/total_composite/std": 0.3263150453567505, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9972518086433411, "sampling/importance_sampling_ratio/min": 0.005771172232925892, "sampling/sampling_logp_difference/max": 5.154880046844482, "sampling/sampling_logp_difference/mean": 0.07868395745754242, "step": 767 }, { "clip_ratio/high_max": 0.0022935778833925724, "clip_ratio/high_mean": 0.0022935778833925724, "clip_ratio/low_mean": 0.0059091257862746716, "clip_ratio/low_min": 0.0059091257862746716, "clip_ratio/region_mean": 0.008202703669667244, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 107.875, "completions/mean_terminated_length": 107.875, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.09003173373639584, "epoch": 0.03084709001084468, "frac_reward_zero_std": 0.0, "grad_norm": 2.0709922313690186, "learning_rate": 7.675757575757577e-06, "loss": -0.0095, "num_tokens": 1694344.0, "reward": 0.5636348724365234, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7045435905456543, "reward_meter_std": 0.13543730974197388, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.10834983736276627, "reward_total_composite_mean": 0.5636348724365234, "reward_total_composite_std": 0.10834985971450806, "reward_total_mean": 0.5636348724365234, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7045435905456543, "rewards/meter/std": 0.13543730974197388, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5636348724365234, "rewards/total_composite/std": 0.10834985971450806, "sampling/importance_sampling_ratio/max": 1.4763460159301758, "sampling/importance_sampling_ratio/mean": 1.0031026601791382, "sampling/importance_sampling_ratio/min": 0.38304516673088074, "sampling/sampling_logp_difference/max": 0.9596023559570312, "sampling/sampling_logp_difference/mean": 0.010335095226764679, "step": 768 }, { "clip_ratio/high_max": 0.0102224723668769, "clip_ratio/high_mean": 0.0102224723668769, "clip_ratio/low_mean": 0.009834110038354993, "clip_ratio/low_min": 0.009834110038354993, "clip_ratio/region_mean": 0.020056582405231893, "completions/clipped_ratio": 0.0, "completions/max_length": 172.0, "completions/max_terminated_length": 172.0, "completions/mean_length": 143.625, "completions/mean_terminated_length": 143.625, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.0793102509342134, "epoch": 0.030887255492629634, "frac_reward_zero_std": 0.0, "grad_norm": 7.707916736602783, "learning_rate": 7.672727272727273e-06, "loss": 0.0714, "num_tokens": 1696973.0, "reward": 0.4105571210384369, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_meter_mean": 0.7441333532333374, "reward_meter_std": 0.4554755687713623, "reward_repeat_penalty_mean": 0.6174242496490479, "reward_repeat_penalty_std": 0.1036364957690239, "reward_std": 0.2803230583667755, "reward_total_composite_mean": 0.4105571210384369, "reward_total_composite_std": 0.2803230583667755, "reward_total_mean": 0.4105571210384369, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/meter/mean": 0.7441333532333374, "rewards/meter/std": 0.4554755687713623, "rewards/repeat_penalty/mean": 0.6174242496490479, "rewards/repeat_penalty/std": 0.1036364957690239, "rewards/total_composite/mean": 0.4105571210384369, "rewards/total_composite/std": 0.2803230583667755, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9966462850570679, "sampling/importance_sampling_ratio/min": 0.0077336556278169155, "sampling/sampling_logp_difference/max": 4.862173557281494, "sampling/sampling_logp_difference/mean": 0.03509281575679779, "step": 769 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.030927420974414588, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 7.66969696969697e-06, "loss": 0.0, "num_tokens": 1698965.0, "reward": 0.3744324743747711, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7828947305679321, "reward_count_adherence_std": 0.03373000770807266, "reward_meter_mean": 0.8939934968948364, "reward_meter_std": 0.13721102476119995, "reward_repeat_penalty_mean": 0.5402884483337402, "reward_repeat_penalty_std": 0.2157072126865387, "reward_std": 0.17441345751285553, "reward_total_composite_mean": 0.3744324743747711, "reward_total_composite_std": 0.17441345751285553, "reward_total_mean": 0.3744324743747711, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7828947305679321, "rewards/count_adherence/std": 0.03373000770807266, "rewards/meter/mean": 0.8939934968948364, "rewards/meter/std": 0.13721102476119995, "rewards/repeat_penalty/mean": 0.5402884483337402, "rewards/repeat_penalty/std": 0.2157072126865387, "rewards/total_composite/mean": 0.3744324743747711, "rewards/total_composite/std": 0.17441345751285553, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 770 }, { "clip_ratio/high_max": 0.010813763190526515, "clip_ratio/high_mean": 0.010813763190526515, "clip_ratio/low_mean": 0.004294316866435111, "clip_ratio/low_min": 0.004294316866435111, "clip_ratio/region_mean": 0.015108080056961626, "completions/clipped_ratio": 0.0, "completions/max_length": 263.0, "completions/max_terminated_length": 263.0, "completions/mean_length": 230.5, "completions/mean_terminated_length": 230.5, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "entropy": 0.13555688876658678, "epoch": 0.030967586456199542, "frac_reward_zero_std": 0.0, "grad_norm": 4.9908342361450195, "learning_rate": 7.666666666666667e-06, "loss": 0.0526, "num_tokens": 1702409.0, "reward": 0.3491661250591278, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.7852883338928223, "reward_meter_std": 0.3382474482059479, "reward_repeat_penalty_mean": 0.5688130855560303, "reward_repeat_penalty_std": 0.20688475668430328, "reward_std": 0.22600221633911133, "reward_total_composite_mean": 0.3491661250591278, "reward_total_composite_std": 0.22600221633911133, "reward_total_mean": 0.3491661250591278, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.7852883338928223, "rewards/meter/std": 0.3382474482059479, "rewards/repeat_penalty/mean": 0.5688130855560303, "rewards/repeat_penalty/std": 0.20688475668430328, "rewards/total_composite/mean": 0.3491661250591278, "rewards/total_composite/std": 0.22600221633911133, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0030508041381836, "sampling/importance_sampling_ratio/min": 0.21555320918560028, "sampling/sampling_logp_difference/max": 1.5345475673675537, "sampling/sampling_logp_difference/mean": 0.018919790163636208, "step": 771 }, { "clip_ratio/high_max": 0.006097560748457909, "clip_ratio/high_mean": 0.006097560748457909, "clip_ratio/low_mean": 0.015005706925876439, "clip_ratio/low_min": 0.015005706925876439, "clip_ratio/region_mean": 0.021103267674334347, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 99.375, "completions/mean_terminated_length": 99.375, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.08717024885118008, "epoch": 0.031007751937984496, "frac_reward_zero_std": 0.0, "grad_norm": 5.677700042724609, "learning_rate": 7.663636363636364e-06, "loss": -0.1232, "num_tokens": 1704516.0, "reward": 0.6195021271705627, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.1157275140285492, "reward_meter_mean": 0.9852504730224609, "reward_meter_std": 0.008375532925128937, "reward_repeat_penalty_mean": 0.7785714864730835, "reward_repeat_penalty_std": 0.039677999913692474, "reward_std": 0.05528085306286812, "reward_total_composite_mean": 0.6195021271705627, "reward_total_composite_std": 0.05528084933757782, "reward_total_mean": 0.6195021271705627, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.1157275140285492, "rewards/meter/mean": 0.9852504730224609, "rewards/meter/std": 0.008375532925128937, "rewards/repeat_penalty/mean": 0.7785714864730835, "rewards/repeat_penalty/std": 0.039677999913692474, "rewards/total_composite/mean": 0.6195021271705627, "rewards/total_composite/std": 0.05528084933757782, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0006357431411743, "sampling/importance_sampling_ratio/min": 0.2092006951570511, "sampling/sampling_logp_difference/max": 1.5644612312316895, "sampling/sampling_logp_difference/mean": 0.020249057561159134, "step": 772 }, { "clip_ratio/high_max": 0.008041649358347058, "clip_ratio/high_mean": 0.008041649358347058, "clip_ratio/low_mean": 0.003349777136463672, "clip_ratio/low_min": 0.003349777136463672, "clip_ratio/region_mean": 0.01139142649481073, "completions/clipped_ratio": 0.0, "completions/max_length": 460.0, "completions/max_terminated_length": 460.0, "completions/mean_length": 409.625, "completions/mean_terminated_length": 409.625, "completions/min_length": 346.0, "completions/min_terminated_length": 346.0, "entropy": 0.07645870675332844, "epoch": 0.03104791741976945, "frac_reward_zero_std": 0.0, "grad_norm": 5.519010066986084, "learning_rate": 7.660606060606062e-06, "loss": 0.0245, "num_tokens": 1709201.0, "reward": 0.17074038088321686, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.3571428656578064, "reward_count_adherence_std": 0.07636035233736038, "reward_meter_mean": 0.9907213449478149, "reward_meter_std": 0.006963523104786873, "reward_repeat_penalty_mean": 0.45148104429244995, "reward_repeat_penalty_std": 0.23929926753044128, "reward_std": 0.10653632134199142, "reward_total_composite_mean": 0.17074038088321686, "reward_total_composite_std": 0.10653632879257202, "reward_total_mean": 0.17074038088321686, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.3571428656578064, "rewards/count_adherence/std": 0.07636035233736038, "rewards/meter/mean": 0.9907213449478149, "rewards/meter/std": 0.006963523104786873, "rewards/repeat_penalty/mean": 0.45148104429244995, "rewards/repeat_penalty/std": 0.23929926753044128, "rewards/total_composite/mean": 0.17074038088321686, "rewards/total_composite/std": 0.10653632879257202, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9987974166870117, "sampling/importance_sampling_ratio/min": 0.006077729165554047, "sampling/sampling_logp_difference/max": 5.103124141693115, "sampling/sampling_logp_difference/mean": 0.02135242149233818, "step": 773 }, { "clip_ratio/high_max": 0.022313665016554296, "clip_ratio/high_mean": 0.022313665016554296, "clip_ratio/low_mean": 0.005654420121572912, "clip_ratio/low_min": 0.005654420121572912, "clip_ratio/region_mean": 0.027968085138127208, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.37627044692635536, "epoch": 0.031088082901554404, "frac_reward_zero_std": 0.0, "grad_norm": 6.2788872718811035, "learning_rate": 7.657575757575757e-06, "loss": -0.0134, "num_tokens": 1711058.0, "reward": 0.7258920669555664, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7258920669555664, "reward_meter_std": 0.40674278140068054, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.40674278140068054, "reward_total_composite_mean": 0.7258920669555664, "reward_total_composite_std": 0.40674278140068054, "reward_total_mean": 0.7258920669555664, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7258920669555664, "rewards/meter/std": 0.40674278140068054, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7258920669555664, "rewards/total_composite/std": 0.40674278140068054, "sampling/importance_sampling_ratio/max": 1.882003903388977, "sampling/importance_sampling_ratio/mean": 1.0082036256790161, "sampling/importance_sampling_ratio/min": 0.1066029891371727, "sampling/sampling_logp_difference/max": 2.2386436462402344, "sampling/sampling_logp_difference/mean": 0.05154965817928314, "step": 774 }, { "clip_ratio/high_max": 0.007139897206798196, "clip_ratio/high_mean": 0.007139897206798196, "clip_ratio/low_mean": 0.00418766331858933, "clip_ratio/low_min": 0.00418766331858933, "clip_ratio/region_mean": 0.011327560525387526, "completions/clipped_ratio": 0.0, "completions/max_length": 188.0, "completions/max_terminated_length": 188.0, "completions/mean_length": 168.375, "completions/mean_terminated_length": 168.375, "completions/min_length": 147.0, "completions/min_terminated_length": 147.0, "entropy": 0.06733021000400186, "epoch": 0.031128248383339358, "frac_reward_zero_std": 0.0, "grad_norm": 2.807114601135254, "learning_rate": 7.654545454545456e-06, "loss": 0.0636, "num_tokens": 1713733.0, "reward": 0.5430476665496826, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_meter_mean": 0.9743660688400269, "reward_meter_std": 0.04519447684288025, "reward_repeat_penalty_mean": 0.6060605645179749, "reward_repeat_penalty_std": 0.13551926612854004, "reward_std": 0.16461656987667084, "reward_total_composite_mean": 0.5430476665496826, "reward_total_composite_std": 0.16461656987667084, "reward_total_mean": 0.5430476665496826, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/meter/mean": 0.9743660688400269, "rewards/meter/std": 0.04519447684288025, "rewards/repeat_penalty/mean": 0.6060605645179749, "rewards/repeat_penalty/std": 0.13551926612854004, "rewards/total_composite/mean": 0.5430476665496826, "rewards/total_composite/std": 0.16461656987667084, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9989210367202759, "sampling/importance_sampling_ratio/min": 3.2826496862981003e-06, "sampling/sampling_logp_difference/max": 12.626859664916992, "sampling/sampling_logp_difference/mean": 0.026703810319304466, "step": 775 }, { "clip_ratio/high_max": 0.009050324792042375, "clip_ratio/high_mean": 0.009050324792042375, "clip_ratio/low_mean": 0.009263938991352916, "clip_ratio/low_min": 0.009263938991352916, "clip_ratio/region_mean": 0.01831426378339529, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 54.5, "completions/mean_terminated_length": 54.5, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.09682174911722541, "epoch": 0.03116841386512431, "frac_reward_zero_std": 0.0, "grad_norm": 7.4227614402771, "learning_rate": 7.651515151515152e-06, "loss": 0.0012, "num_tokens": 1715409.0, "reward": 0.9904471635818481, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9904471635818481, "reward_meter_std": 0.002613567281514406, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002613575430586934, "reward_total_composite_mean": 0.9904471635818481, "reward_total_composite_std": 0.002613567281514406, "reward_total_mean": 0.9904471635818481, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9904471635818481, "rewards/meter/std": 0.002613567281514406, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9904471635818481, "rewards/total_composite/std": 0.002613567281514406, "sampling/importance_sampling_ratio/max": 1.4630638360977173, "sampling/importance_sampling_ratio/mean": 0.9967302083969116, "sampling/importance_sampling_ratio/min": 0.3387613296508789, "sampling/sampling_logp_difference/max": 1.0824594497680664, "sampling/sampling_logp_difference/mean": 0.024185476824641228, "step": 776 }, { "clip_ratio/high_max": 0.013795045204460621, "clip_ratio/high_mean": 0.013795045204460621, "clip_ratio/low_mean": 0.013795045437291265, "clip_ratio/low_min": 0.013795045437291265, "clip_ratio/region_mean": 0.027590090641751885, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 36.5, "completions/mean_terminated_length": 36.5, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.11897748988121748, "epoch": 0.031208579346909265, "frac_reward_zero_std": 0.0, "grad_norm": 8.215063095092773, "learning_rate": 7.648484848484849e-06, "loss": 0.0046, "num_tokens": 1716925.0, "reward": 0.978127121925354, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.978127121925354, "reward_meter_std": 0.0063257296569645405, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006325736176222563, "reward_total_composite_mean": 0.978127121925354, "reward_total_composite_std": 0.0063257296569645405, "reward_total_mean": 0.978127121925354, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.978127121925354, "rewards/meter/std": 0.0063257296569645405, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.978127121925354, "rewards/total_composite/std": 0.0063257296569645405, "sampling/importance_sampling_ratio/max": 1.6868826150894165, "sampling/importance_sampling_ratio/mean": 1.0036948919296265, "sampling/importance_sampling_ratio/min": 0.23150713741779327, "sampling/sampling_logp_difference/max": 1.4631445407867432, "sampling/sampling_logp_difference/mean": 0.03158137574791908, "step": 777 }, { "clip_ratio/high_max": 0.020928236190229654, "clip_ratio/high_mean": 0.020928236190229654, "clip_ratio/low_mean": 0.02250340231694281, "clip_ratio/low_min": 0.02250340231694281, "clip_ratio/region_mean": 0.043431638507172465, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 72.25, "completions/mean_terminated_length": 72.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.25925541296601295, "epoch": 0.03124874482869422, "frac_reward_zero_std": 0.0, "grad_norm": 5.06115198135376, "learning_rate": 7.645454545454546e-06, "loss": 0.014, "num_tokens": 1718951.0, "reward": 0.9873093366622925, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9873093366622925, "reward_meter_std": 0.01036460418254137, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010364595800638199, "reward_total_composite_mean": 0.9873093366622925, "reward_total_composite_std": 0.01036460418254137, "reward_total_mean": 0.9873093366622925, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9873093366622925, "rewards/meter/std": 0.01036460418254137, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9873093366622925, "rewards/total_composite/std": 0.01036460418254137, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0149582624435425, "sampling/importance_sampling_ratio/min": 0.4104529917240143, "sampling/sampling_logp_difference/max": 0.8904938697814941, "sampling/sampling_logp_difference/mean": 0.03534568101167679, "step": 778 }, { "clip_ratio/high_max": 0.030813875840976834, "clip_ratio/high_mean": 0.030813875840976834, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/region_mean": 0.039375519612804055, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 182.5, "completions/mean_terminated_length": 72.66667175292969, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.38310878723859787, "epoch": 0.03128891031047917, "frac_reward_zero_std": 0.0, "grad_norm": 1.4244983196258545, "learning_rate": 7.642424242424244e-06, "loss": -0.1238, "num_tokens": 1720483.0, "reward": 0.5616785287857056, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6190794110298157, "reward_meter_std": 0.3674232065677643, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.43257129192352295, "reward_total_composite_mean": 0.5616785287857056, "reward_total_composite_std": 0.43257132172584534, "reward_total_mean": 0.5616785287857056, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6190794110298157, "rewards/meter/std": 0.3674232065677643, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5616785287857056, "rewards/total_composite/std": 0.43257132172584534, "sampling/importance_sampling_ratio/max": 1.9660524129867554, "sampling/importance_sampling_ratio/mean": 1.0154547691345215, "sampling/importance_sampling_ratio/min": 0.2951788604259491, "sampling/sampling_logp_difference/max": 1.2201738357543945, "sampling/sampling_logp_difference/mean": 0.07225891947746277, "step": 779 }, { "clip_ratio/high_max": 0.029140884289518, "clip_ratio/high_mean": 0.029140884289518, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/region_mean": 0.0339485767763108, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 182.0, "completions/mean_length": 174.375, "completions/mean_terminated_length": 126.14286041259766, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.24183030799031258, "epoch": 0.03132907579226413, "frac_reward_zero_std": 0.0, "grad_norm": 3.119354009628296, "learning_rate": 7.639393939393939e-06, "loss": -0.0498, "num_tokens": 1722862.0, "reward": 0.4575069546699524, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.6168656349182129, "reward_meter_std": 0.4164867103099823, "reward_repeat_penalty_mean": 0.7492559552192688, "reward_repeat_penalty_std": 0.14768067002296448, "reward_std": 0.3478102684020996, "reward_total_composite_mean": 0.4575069546699524, "reward_total_composite_std": 0.3478102684020996, "reward_total_mean": 0.4575069546699524, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.6168656349182129, "rewards/meter/std": 0.4164867103099823, "rewards/repeat_penalty/mean": 0.7492559552192688, "rewards/repeat_penalty/std": 0.14768067002296448, "rewards/total_composite/mean": 0.4575069546699524, "rewards/total_composite/std": 0.3478102684020996, "sampling/importance_sampling_ratio/max": 1.7341415882110596, "sampling/importance_sampling_ratio/mean": 1.0024807453155518, "sampling/importance_sampling_ratio/min": 0.34019285440444946, "sampling/sampling_logp_difference/max": 1.078242540359497, "sampling/sampling_logp_difference/mean": 0.03274521231651306, "step": 780 }, { "clip_ratio/high_max": 0.016590908635407686, "clip_ratio/high_mean": 0.016590908635407686, "clip_ratio/low_mean": 0.04398810095153749, "clip_ratio/low_min": 0.04398810095153749, "clip_ratio/region_mean": 0.060579009586945176, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 47.375, "completions/mean_terminated_length": 47.375, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.9118200056254864, "epoch": 0.03136924127404908, "frac_reward_zero_std": 0.0, "grad_norm": 12.43602180480957, "learning_rate": 7.636363636363638e-06, "loss": -0.0116, "num_tokens": 1724537.0, "reward": 0.4856577217578888, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5619980096817017, "reward_meter_std": 0.3257886469364166, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.3285104036331177, "reward_total_composite_mean": 0.4856577217578888, "reward_total_composite_std": 0.3285104036331177, "reward_total_mean": 0.4856577217578888, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5619980096817017, "rewards/meter/std": 0.3257886469364166, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.4856577217578888, "rewards/total_composite/std": 0.3285104036331177, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9970747232437134, "sampling/importance_sampling_ratio/min": 0.18346960842609406, "sampling/sampling_logp_difference/max": 1.6957062482833862, "sampling/sampling_logp_difference/mean": 0.09212920814752579, "step": 781 }, { "clip_ratio/high_max": 0.013982121949084103, "clip_ratio/high_mean": 0.013982121949084103, "clip_ratio/low_mean": 0.008460594224743545, "clip_ratio/low_min": 0.008460594224743545, "clip_ratio/region_mean": 0.022442716173827648, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 166.75, "completions/mean_terminated_length": 117.42857360839844, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.2663711039349437, "epoch": 0.031409406755834035, "frac_reward_zero_std": 0.0, "grad_norm": 1.6532626152038574, "learning_rate": 7.633333333333334e-06, "loss": -0.1573, "num_tokens": 1726623.0, "reward": 0.5303109288215637, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6895759105682373, "reward_meter_std": 0.26801931858062744, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.19820624589920044, "reward_std": 0.28132033348083496, "reward_total_composite_mean": 0.5303109288215637, "reward_total_composite_std": 0.28132033348083496, "reward_total_mean": 0.5303109288215637, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6895759105682373, "rewards/meter/std": 0.26801931858062744, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.19820624589920044, "rewards/total_composite/mean": 0.5303109288215637, "rewards/total_composite/std": 0.28132033348083496, "sampling/importance_sampling_ratio/max": 1.9478799104690552, "sampling/importance_sampling_ratio/mean": 1.006204605102539, "sampling/importance_sampling_ratio/min": 0.31285813450813293, "sampling/sampling_logp_difference/max": 1.1620054244995117, "sampling/sampling_logp_difference/mean": 0.03518048673868179, "step": 782 }, { "clip_ratio/high_max": 0.013986280770041049, "clip_ratio/high_mean": 0.013986280770041049, "clip_ratio/low_mean": 0.017455301131121814, "clip_ratio/low_min": 0.017455301131121814, "clip_ratio/region_mean": 0.03144158190116286, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 79.5, "completions/mean_terminated_length": 79.5, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.3720816671848297, "epoch": 0.03144957223761899, "frac_reward_zero_std": 0.0, "grad_norm": 4.994919776916504, "learning_rate": 7.630303030303031e-06, "loss": 0.0077, "num_tokens": 1728611.0, "reward": 0.6204426288604736, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6204426288604736, "reward_meter_std": 0.3357177972793579, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3357177972793579, "reward_total_composite_mean": 0.6204426288604736, "reward_total_composite_std": 0.3357177972793579, "reward_total_mean": 0.6204426288604736, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6204426288604736, "rewards/meter/std": 0.3357177972793579, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6204426288604736, "rewards/total_composite/std": 0.3357177972793579, "sampling/importance_sampling_ratio/max": 1.7397865056991577, "sampling/importance_sampling_ratio/mean": 1.011016845703125, "sampling/importance_sampling_ratio/min": 0.34513458609580994, "sampling/sampling_logp_difference/max": 1.0638208389282227, "sampling/sampling_logp_difference/mean": 0.04477819800376892, "step": 783 }, { "clip_ratio/high_max": 0.012099922401830554, "clip_ratio/high_mean": 0.012099922401830554, "clip_ratio/low_mean": 0.018428327050060034, "clip_ratio/low_min": 0.018428327050060034, "clip_ratio/region_mean": 0.030528249451890588, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.375, "completions/mean_terminated_length": 61.375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2184693105518818, "epoch": 0.03148973771940394, "frac_reward_zero_std": 0.0, "grad_norm": 7.485208988189697, "learning_rate": 7.627272727272727e-06, "loss": 0.0052, "num_tokens": 1730350.0, "reward": 0.453016996383667, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5055776834487915, "reward_meter_std": 0.32300832867622375, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_std": 0.33306846022605896, "reward_total_composite_mean": 0.453016996383667, "reward_total_composite_std": 0.33306846022605896, "reward_total_mean": 0.453016996383667, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5055776834487915, "rewards/meter/std": 0.32300832867622375, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.453016996383667, "rewards/total_composite/std": 0.33306846022605896, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9948754906654358, "sampling/importance_sampling_ratio/min": 0.17718835175037384, "sampling/sampling_logp_difference/max": 1.7305419445037842, "sampling/sampling_logp_difference/mean": 0.047304026782512665, "step": 784 }, { "clip_ratio/high_max": 0.01767290150746703, "clip_ratio/high_mean": 0.01767290150746703, "clip_ratio/low_mean": 0.010869022691622376, "clip_ratio/low_min": 0.010869022691622376, "clip_ratio/region_mean": 0.028541924199089408, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 104.75, "completions/mean_terminated_length": 104.75, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.31474705785512924, "epoch": 0.0315299032011889, "frac_reward_zero_std": 0.0, "grad_norm": 6.059953212738037, "learning_rate": 7.6242424242424254e-06, "loss": 0.0036, "num_tokens": 1732524.0, "reward": 0.46458256244659424, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.49319469928741455, "reward_meter_std": 0.2711309790611267, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.2836271822452545, "reward_total_composite_mean": 0.46458256244659424, "reward_total_composite_std": 0.2836271822452545, "reward_total_mean": 0.46458256244659424, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.49319469928741455, "rewards/meter/std": 0.2711309790611267, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.46458256244659424, "rewards/total_composite/std": 0.2836271822452545, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0086406469345093, "sampling/importance_sampling_ratio/min": 0.27668195962905884, "sampling/sampling_logp_difference/max": 1.2848865985870361, "sampling/sampling_logp_difference/mean": 0.042036473751068115, "step": 785 }, { "clip_ratio/high_max": 0.0042421949619892985, "clip_ratio/high_mean": 0.0042421949619892985, "clip_ratio/low_mean": 0.004607091657817364, "clip_ratio/low_min": 0.004607091657817364, "clip_ratio/region_mean": 0.008849286619806662, "completions/clipped_ratio": 0.0, "completions/max_length": 406.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 368.25, "completions/mean_terminated_length": 368.25, "completions/min_length": 322.0, "completions/min_terminated_length": 322.0, "entropy": 0.07080868305638433, "epoch": 0.03157006868297385, "frac_reward_zero_std": 0.0, "grad_norm": 2.024092435836792, "learning_rate": 7.621212121212122e-06, "loss": -0.0567, "num_tokens": 1737190.0, "reward": 0.3579794764518738, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7014279365539551, "reward_meter_std": 0.43751856684684753, "reward_repeat_penalty_mean": 0.5806276798248291, "reward_repeat_penalty_std": 0.06668182462453842, "reward_std": 0.22696760296821594, "reward_total_composite_mean": 0.3579794764518738, "reward_total_composite_std": 0.22696760296821594, "reward_total_mean": 0.3579794764518738, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7014279365539551, "rewards/meter/std": 0.43751856684684753, "rewards/repeat_penalty/mean": 0.5806276798248291, "rewards/repeat_penalty/std": 0.06668182462453842, "rewards/total_composite/mean": 0.3579794764518738, "rewards/total_composite/std": 0.22696760296821594, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0009483098983765, "sampling/importance_sampling_ratio/min": 0.2249726504087448, "sampling/sampling_logp_difference/max": 1.491776466369629, "sampling/sampling_logp_difference/mean": 0.012620110996067524, "step": 786 }, { "clip_ratio/high_max": 0.007280809339135885, "clip_ratio/high_mean": 0.007280809339135885, "clip_ratio/low_mean": 0.01558087719604373, "clip_ratio/low_min": 0.01558087719604373, "clip_ratio/region_mean": 0.022861686535179615, "completions/clipped_ratio": 0.0, "completions/max_length": 162.0, "completions/max_terminated_length": 162.0, "completions/mean_length": 153.75, "completions/mean_terminated_length": 153.75, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.11316746287047863, "epoch": 0.031610234164758805, "frac_reward_zero_std": 0.0, "grad_norm": 3.6839067935943604, "learning_rate": 7.618181818181819e-06, "loss": 0.0123, "num_tokens": 1739852.0, "reward": 0.435272216796875, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7929247617721558, "reward_meter_std": 0.10911522805690765, "reward_repeat_penalty_mean": 0.6818181872367859, "reward_repeat_penalty_std": 0.06872082501649857, "reward_std": 0.09194309264421463, "reward_total_composite_mean": 0.435272216796875, "reward_total_composite_std": 0.09194310009479523, "reward_total_mean": 0.435272216796875, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7929247617721558, "rewards/meter/std": 0.10911522805690765, "rewards/repeat_penalty/mean": 0.6818181872367859, "rewards/repeat_penalty/std": 0.06872082501649857, "rewards/total_composite/mean": 0.435272216796875, "rewards/total_composite/std": 0.09194310009479523, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9994563460350037, "sampling/importance_sampling_ratio/min": 0.14561693370342255, "sampling/sampling_logp_difference/max": 1.9267759323120117, "sampling/sampling_logp_difference/mean": 0.028897186741232872, "step": 787 }, { "clip_ratio/high_max": 0.01074189692735672, "clip_ratio/high_mean": 0.01074189692735672, "clip_ratio/low_mean": 0.012860541231930256, "clip_ratio/low_min": 0.012860541231930256, "clip_ratio/region_mean": 0.023602438159286976, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.16343743726611137, "epoch": 0.03165039964654376, "frac_reward_zero_std": 0.0, "grad_norm": 7.856208324432373, "learning_rate": 7.6151515151515155e-06, "loss": -0.0051, "num_tokens": 1741588.0, "reward": 0.6469426155090332, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6469426155090332, "reward_meter_std": 0.24868278205394745, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.24868276715278625, "reward_total_composite_mean": 0.6469426155090332, "reward_total_composite_std": 0.24868278205394745, "reward_total_mean": 0.6469426155090332, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6469426155090332, "rewards/meter/std": 0.24868278205394745, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6469426155090332, "rewards/total_composite/std": 0.24868278205394745, "sampling/importance_sampling_ratio/max": 1.8282591104507446, "sampling/importance_sampling_ratio/mean": 1.0034902095794678, "sampling/importance_sampling_ratio/min": 0.5131197571754456, "sampling/sampling_logp_difference/max": 0.6672461032867432, "sampling/sampling_logp_difference/mean": 0.022744039073586464, "step": 788 }, { "clip_ratio/high_max": 0.008072916883975267, "clip_ratio/high_mean": 0.008072916883975267, "clip_ratio/low_mean": 0.008072916883975267, "clip_ratio/low_min": 0.008072916883975267, "clip_ratio/region_mean": 0.016145833767950535, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 31.25, "completions/mean_terminated_length": 31.25, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.17746192403137684, "epoch": 0.03169056512832871, "frac_reward_zero_std": 0.0, "grad_norm": 13.922452926635742, "learning_rate": 7.612121212121213e-06, "loss": 0.0131, "num_tokens": 1743054.0, "reward": 0.9813005924224854, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9813005924224854, "reward_meter_std": 0.016189413145184517, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.016189415007829666, "reward_total_composite_mean": 0.9813005924224854, "reward_total_composite_std": 0.016189413145184517, "reward_total_mean": 0.9813005924224854, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9813005924224854, "rewards/meter/std": 0.016189413145184517, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9813005924224854, "rewards/total_composite/std": 0.016189413145184517, "sampling/importance_sampling_ratio/max": 1.8372166156768799, "sampling/importance_sampling_ratio/mean": 0.9976152181625366, "sampling/importance_sampling_ratio/min": 0.3405851721763611, "sampling/sampling_logp_difference/max": 1.0770900249481201, "sampling/sampling_logp_difference/mean": 0.033597491681575775, "step": 789 }, { "clip_ratio/high_max": 0.019598963437601924, "clip_ratio/high_mean": 0.019598963437601924, "clip_ratio/low_mean": 0.009868005756288767, "clip_ratio/low_min": 0.009868005756288767, "clip_ratio/region_mean": 0.02946696919389069, "completions/clipped_ratio": 0.0, "completions/max_length": 92.0, "completions/max_terminated_length": 92.0, "completions/mean_length": 89.625, "completions/mean_terminated_length": 89.625, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.11805877834558487, "epoch": 0.03173073061011367, "frac_reward_zero_std": 0.0, "grad_norm": 6.977390766143799, "learning_rate": 7.609090909090909e-06, "loss": -0.0039, "num_tokens": 1745187.0, "reward": 0.6789125204086304, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8486406207084656, "reward_meter_std": 0.2562169134616852, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.20497353374958038, "reward_total_composite_mean": 0.6789125204086304, "reward_total_composite_std": 0.20497353374958038, "reward_total_mean": 0.6789125204086304, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8486406207084656, "rewards/meter/std": 0.2562169134616852, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6789125204086304, "rewards/total_composite/std": 0.20497353374958038, "sampling/importance_sampling_ratio/max": 1.8353419303894043, "sampling/importance_sampling_ratio/mean": 1.0061084032058716, "sampling/importance_sampling_ratio/min": 0.44276192784309387, "sampling/sampling_logp_difference/max": 0.814723014831543, "sampling/sampling_logp_difference/mean": 0.021279064938426018, "step": 790 }, { "clip_ratio/high_max": 0.00779381615575403, "clip_ratio/high_mean": 0.00779381615575403, "clip_ratio/low_mean": 0.0032260402804240584, "clip_ratio/low_min": 0.0032260402804240584, "clip_ratio/region_mean": 0.011019856436178088, "completions/clipped_ratio": 0.0, "completions/max_length": 258.0, "completions/max_terminated_length": 258.0, "completions/mean_length": 247.25, "completions/mean_terminated_length": 247.25, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.042171002831310034, "epoch": 0.03177089609189862, "frac_reward_zero_std": 0.0, "grad_norm": 1.8626375198364258, "learning_rate": 7.606060606060606e-06, "loss": -0.0395, "num_tokens": 1748773.0, "reward": 0.4338645935058594, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.9959322214126587, "reward_meter_std": 0.004204806871712208, "reward_repeat_penalty_mean": 0.49047619104385376, "reward_repeat_penalty_std": 0.17105023562908173, "reward_std": 0.12500372529029846, "reward_total_composite_mean": 0.4338645935058594, "reward_total_composite_std": 0.12500374019145966, "reward_total_mean": 0.4338645935058594, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.9959322214126587, "rewards/meter/std": 0.004204806871712208, "rewards/repeat_penalty/mean": 0.49047619104385376, "rewards/repeat_penalty/std": 0.17105023562908173, "rewards/total_composite/mean": 0.4338645935058594, "rewards/total_composite/std": 0.12500374019145966, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0004253387451172, "sampling/importance_sampling_ratio/min": 0.0588078573346138, "sampling/sampling_logp_difference/max": 2.833479881286621, "sampling/sampling_logp_difference/mean": 0.014479896053671837, "step": 791 }, { "clip_ratio/high_max": 0.00658258656039834, "clip_ratio/high_mean": 0.00658258656039834, "clip_ratio/low_mean": 0.010955466306768358, "clip_ratio/low_min": 0.010955466306768358, "clip_ratio/region_mean": 0.017538052867166698, "completions/clipped_ratio": 0.0, "completions/max_length": 148.0, "completions/max_terminated_length": 148.0, "completions/mean_length": 135.0, "completions/mean_terminated_length": 135.0, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.0634374669753015, "epoch": 0.031811061573683574, "frac_reward_zero_std": 0.0, "grad_norm": 3.4738333225250244, "learning_rate": 7.603030303030303e-06, "loss": 0.023, "num_tokens": 1751485.0, "reward": 0.6611686944961548, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.991753101348877, "reward_meter_std": 0.002629074966534972, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0017527244053781033, "reward_total_composite_mean": 0.6611686944961548, "reward_total_composite_std": 0.0017527303425595164, "reward_total_mean": 0.6611686944961548, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.991753101348877, "rewards/meter/std": 0.002629074966534972, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6611686944961548, "rewards/total_composite/std": 0.0017527303425595164, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0068343877792358, "sampling/importance_sampling_ratio/min": 0.13520930707454681, "sampling/sampling_logp_difference/max": 2.0009312629699707, "sampling/sampling_logp_difference/mean": 0.019971484318375587, "step": 792 }, { "clip_ratio/high_max": 0.0335467669647187, "clip_ratio/high_mean": 0.0335467669647187, "clip_ratio/low_mean": 0.006658692145720124, "clip_ratio/low_min": 0.006658692145720124, "clip_ratio/region_mean": 0.040205459110438824, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 56.0, "completions/mean_terminated_length": 56.0, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.15502143744379282, "epoch": 0.03185122705546853, "frac_reward_zero_std": 0.0, "grad_norm": 5.889811038970947, "learning_rate": 7.600000000000001e-06, "loss": 0.0066, "num_tokens": 1753253.0, "reward": 0.9050641655921936, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9871419668197632, "reward_meter_std": 0.005153529345989227, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.15341757237911224, "reward_total_composite_mean": 0.9050641655921936, "reward_total_composite_std": 0.15341757237911224, "reward_total_mean": 0.9050641655921936, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9871419668197632, "rewards/meter/std": 0.005153529345989227, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.9050641655921936, "rewards/total_composite/std": 0.15341757237911224, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9988625645637512, "sampling/importance_sampling_ratio/min": 0.05082041397690773, "sampling/sampling_logp_difference/max": 2.979457139968872, "sampling/sampling_logp_difference/mean": 0.054076679050922394, "step": 793 }, { "clip_ratio/high_max": 0.007044379832223058, "clip_ratio/high_mean": 0.007044379832223058, "clip_ratio/low_mean": 0.019052310031838715, "clip_ratio/low_min": 0.019052310031838715, "clip_ratio/region_mean": 0.026096689864061773, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.625, "completions/mean_terminated_length": 71.625, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.15924948174506426, "epoch": 0.03189139253725348, "frac_reward_zero_std": 0.0, "grad_norm": 6.565411567687988, "learning_rate": 7.596969696969697e-06, "loss": 0.002, "num_tokens": 1755042.0, "reward": 0.9123142957687378, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9123142957687378, "reward_meter_std": 0.09913700073957443, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09913701564073563, "reward_total_composite_mean": 0.9123142957687378, "reward_total_composite_std": 0.09913700073957443, "reward_total_mean": 0.9123142957687378, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9123142957687378, "rewards/meter/std": 0.09913700073957443, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9123142957687378, "rewards/total_composite/std": 0.09913700073957443, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0043187141418457, "sampling/importance_sampling_ratio/min": 0.249431312084198, "sampling/sampling_logp_difference/max": 1.3885717391967773, "sampling/sampling_logp_difference/mean": 0.034922052174806595, "step": 794 }, { "clip_ratio/high_max": 0.0034246575087308884, "clip_ratio/high_mean": 0.0034246575087308884, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/region_mean": 0.01198630128055811, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 73.0, "completions/mean_terminated_length": 73.0, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.07319271843880415, "epoch": 0.031931558019038436, "frac_reward_zero_std": 0.0, "grad_norm": 2.6046133041381836, "learning_rate": 7.593939393939395e-06, "loss": 0.0046, "num_tokens": 1756866.0, "reward": 0.613559365272522, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.613559365272522, "reward_meter_std": 0.046057041734457016, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04605703800916672, "reward_total_composite_mean": 0.613559365272522, "reward_total_composite_std": 0.046057041734457016, "reward_total_mean": 0.613559365272522, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.613559365272522, "rewards/meter/std": 0.046057041734457016, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.613559365272522, "rewards/total_composite/std": 0.046057041734457016, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0046606063842773, "sampling/importance_sampling_ratio/min": 0.5393000841140747, "sampling/sampling_logp_difference/max": 0.9375072717666626, "sampling/sampling_logp_difference/mean": 0.01630816049873829, "step": 795 }, { "clip_ratio/high_max": 0.002516112755984068, "clip_ratio/high_mean": 0.002516112755984068, "clip_ratio/low_mean": 0.0004734848625957966, "clip_ratio/low_min": 0.0004734848625957966, "clip_ratio/region_mean": 0.0029895976185798645, "completions/clipped_ratio": 0.0, "completions/max_length": 264.0, "completions/max_terminated_length": 264.0, "completions/mean_length": 250.125, "completions/mean_terminated_length": 250.125, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.023278776556253433, "epoch": 0.03197172350082339, "frac_reward_zero_std": 0.0, "grad_norm": 1.5290881395339966, "learning_rate": 7.590909090909091e-06, "loss": 0.0182, "num_tokens": 1760371.0, "reward": 0.3849101662635803, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6284428834915161, "reward_meter_std": 0.1288241744041443, "reward_repeat_penalty_mean": 0.6098901033401489, "reward_repeat_penalty_std": 0.015540807507932186, "reward_std": 0.08409524708986282, "reward_total_composite_mean": 0.3849101662635803, "reward_total_composite_std": 0.08409524708986282, "reward_total_mean": 0.3849101662635803, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6284428834915161, "rewards/meter/std": 0.1288241744041443, "rewards/repeat_penalty/mean": 0.6098901033401489, "rewards/repeat_penalty/std": 0.015540807507932186, "rewards/total_composite/mean": 0.3849101662635803, "rewards/total_composite/std": 0.08409524708986282, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0002148151397705, "sampling/importance_sampling_ratio/min": 0.3427882194519043, "sampling/sampling_logp_difference/max": 1.0706424713134766, "sampling/sampling_logp_difference/mean": 0.006375753786414862, "step": 796 }, { "clip_ratio/high_max": 0.009473072132095695, "clip_ratio/high_mean": 0.009473072132095695, "clip_ratio/low_mean": 0.003246753243729472, "clip_ratio/low_min": 0.003246753243729472, "clip_ratio/region_mean": 0.012719825375825167, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 77.5, "completions/mean_terminated_length": 77.5, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.14471599273383617, "epoch": 0.032011888982608344, "frac_reward_zero_std": 0.0, "grad_norm": 8.541836738586426, "learning_rate": 7.587878787878788e-06, "loss": -0.0121, "num_tokens": 1762455.0, "reward": 0.891382098197937, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9952297210693359, "reward_meter_std": 0.002311403863132, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.1964196413755417, "reward_total_composite_mean": 0.891382098197937, "reward_total_composite_std": 0.1964196413755417, "reward_total_mean": 0.891382098197937, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9952297210693359, "rewards/meter/std": 0.002311403863132, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.891382098197937, "rewards/total_composite/std": 0.1964196413755417, "sampling/importance_sampling_ratio/max": 1.482370376586914, "sampling/importance_sampling_ratio/mean": 0.9954510927200317, "sampling/importance_sampling_ratio/min": 0.03526677191257477, "sampling/sampling_logp_difference/max": 3.3448140621185303, "sampling/sampling_logp_difference/mean": 0.034574542194604874, "step": 797 }, { "clip_ratio/high_max": 0.00390625, "clip_ratio/high_mean": 0.00390625, "clip_ratio/low_mean": 0.020607191254384816, "clip_ratio/low_min": 0.020607191254384816, "clip_ratio/region_mean": 0.024513441254384816, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.1323620891198516, "epoch": 0.0320520544643933, "frac_reward_zero_std": 0.0, "grad_norm": 6.7140374183654785, "learning_rate": 7.584848484848486e-06, "loss": 0.0167, "num_tokens": 1764291.0, "reward": 0.6813795566558838, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6813795566558838, "reward_meter_std": 0.04148668423295021, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04148669168353081, "reward_total_composite_mean": 0.6813795566558838, "reward_total_composite_std": 0.04148668423295021, "reward_total_mean": 0.6813795566558838, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6813795566558838, "rewards/meter/std": 0.04148668423295021, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6813795566558838, "rewards/total_composite/std": 0.04148668423295021, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010377049446106, "sampling/importance_sampling_ratio/min": 0.5285674333572388, "sampling/sampling_logp_difference/max": 0.9566974639892578, "sampling/sampling_logp_difference/mean": 0.023791270330548286, "step": 798 }, { "clip_ratio/high_max": 0.002109704539179802, "clip_ratio/high_mean": 0.002109704539179802, "clip_ratio/low_mean": 0.003645833523478359, "clip_ratio/low_min": 0.003645833523478359, "clip_ratio/region_mean": 0.005755538062658161, "completions/clipped_ratio": 0.0, "completions/max_length": 240.0, "completions/max_terminated_length": 240.0, "completions/mean_length": 239.375, "completions/mean_terminated_length": 239.375, "completions/min_length": 237.0, "completions/min_terminated_length": 237.0, "entropy": 0.030161422211676836, "epoch": 0.03209221994617825, "frac_reward_zero_std": 0.0, "grad_norm": 1.2270196676254272, "learning_rate": 7.581818181818183e-06, "loss": 0.0044, "num_tokens": 1767790.0, "reward": 0.38379937410354614, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7462765574455261, "reward_meter_std": 0.09833642095327377, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_std": 0.050573013722896576, "reward_total_composite_mean": 0.38379937410354614, "reward_total_composite_std": 0.05057300627231598, "reward_total_mean": 0.38379937410354614, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7462765574455261, "rewards/meter/std": 0.09833642095327377, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.38379937410354614, "rewards/total_composite/std": 0.05057300627231598, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9996190071105957, "sampling/importance_sampling_ratio/min": 0.07412450760602951, "sampling/sampling_logp_difference/max": 2.6020090579986572, "sampling/sampling_logp_difference/mean": 0.007426128257066011, "step": 799 }, { "clip_ratio/high_max": 0.0143592240056023, "clip_ratio/high_mean": 0.0143592240056023, "clip_ratio/low_mean": 0.024260351667180657, "clip_ratio/low_min": 0.024260351667180657, "clip_ratio/region_mean": 0.03861957567278296, "completions/clipped_ratio": 0.0, "completions/max_length": 130.0, "completions/max_terminated_length": 130.0, "completions/mean_length": 119.375, "completions/mean_terminated_length": 119.375, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.23464529775083065, "epoch": 0.032132385427963206, "frac_reward_zero_std": 0.0, "grad_norm": 4.79632568359375, "learning_rate": 7.57878787878788e-06, "loss": 0.0432, "num_tokens": 1770025.0, "reward": 0.4120614528656006, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5170607566833496, "reward_meter_std": 0.48065322637557983, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.386055052280426, "reward_total_composite_mean": 0.4120614528656006, "reward_total_composite_std": 0.386055052280426, "reward_total_mean": 0.4120614528656006, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5170607566833496, "rewards/meter/std": 0.48065322637557983, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.4120614528656006, "rewards/total_composite/std": 0.386055052280426, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9977750182151794, "sampling/importance_sampling_ratio/min": 0.3169117867946625, "sampling/sampling_logp_difference/max": 1.1491317749023438, "sampling/sampling_logp_difference/mean": 0.04103344678878784, "step": 800 }, { "epoch": 0.032132385427963206, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/max_length": 461.9230769230769, "eval_completions/max_terminated_length": 414.46153846153845, "eval_completions/mean_length": 234.29807692307693, "eval_completions/mean_terminated_length": 216.9835216815655, "eval_completions/min_length": 63.84615384615385, "eval_completions/min_terminated_length": 63.84615384615385, "eval_entropy": 0.047992275741237864, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 1770025.0, "eval_reward": 0.3991968219096844, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.9552615697567279, "eval_reward_count_adherence_std": 0.08013723160211857, "eval_reward_meter_mean": 0.6419616112342248, "eval_reward_meter_std": 0.3754527878302794, "eval_reward_repeat_penalty_mean": 0.6556147245260385, "eval_reward_repeat_penalty_std": 0.24277657327743676, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.3991968219096844, "eval_reward_total_composite_std": 0.3098094039238416, "eval_reward_total_mean": 0.3991968219096844, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.9552615697567279, "eval_rewards/count_adherence/std": 0.08013723160211857, "eval_rewards/meter/mean": 0.6419616112342248, "eval_rewards/meter/std": 0.3754527878302794, "eval_rewards/repeat_penalty/mean": 0.6556147245260385, "eval_rewards/repeat_penalty/std": 0.24277657327743676, "eval_rewards/total_composite/mean": 0.3991968219096844, "eval_rewards/total_composite/std": 0.3098094039238416, "eval_runtime": 85.9157, "eval_samples_per_second": 1.21, "eval_sampling/importance_sampling_ratio/max": 1.2961730773632343, "eval_sampling/importance_sampling_ratio/mean": 1.0009051194557776, "eval_sampling/importance_sampling_ratio/min": 0.5230464408030877, "eval_sampling/sampling_logp_difference/max": 0.7075341939926147, "eval_sampling/sampling_logp_difference/mean": 0.0054089168552309275, "eval_steps_per_second": 0.151, "step": 800 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.018601843621581793, "clip_ratio/low_min": 0.018601843621581793, "clip_ratio/region_mean": 0.018601843621581793, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.08063888642936945, "epoch": 0.03217255090974816, "frac_reward_zero_std": 0.0, "grad_norm": 4.925754070281982, "learning_rate": 7.5757575757575764e-06, "loss": 0.0018, "num_tokens": 1771810.0, "reward": 0.7430859804153442, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7430859804153442, "reward_meter_std": 0.04014641046524048, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.040146395564079285, "reward_total_composite_mean": 0.7430859804153442, "reward_total_composite_std": 0.04014641046524048, "reward_total_mean": 0.7430859804153442, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7430859804153442, "rewards/meter/std": 0.04014641046524048, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7430859804153442, "rewards/total_composite/std": 0.04014641046524048, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0070921182632446, "sampling/importance_sampling_ratio/min": 0.33353158831596375, "sampling/sampling_logp_difference/max": 1.098017692565918, "sampling/sampling_logp_difference/mean": 0.02094501443207264, "step": 801 }, { "clip_ratio/high_max": 0.00995732587762177, "clip_ratio/high_mean": 0.00995732587762177, "clip_ratio/low_mean": 0.009703947464004159, "clip_ratio/low_min": 0.009703947464004159, "clip_ratio/region_mean": 0.01966127334162593, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 37.625, "completions/mean_terminated_length": 37.625, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.1240494973026216, "epoch": 0.032212716391533114, "frac_reward_zero_std": 0.0, "grad_norm": 7.339356422424316, "learning_rate": 7.572727272727274e-06, "loss": 0.0172, "num_tokens": 1773359.0, "reward": 0.13143374025821686, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.13143374025821686, "reward_meter_std": 0.11004206538200378, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.11004206538200378, "reward_total_composite_mean": 0.13143374025821686, "reward_total_composite_std": 0.11004206538200378, "reward_total_mean": 0.13143374025821686, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.13143374025821686, "rewards/meter/std": 0.11004206538200378, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.13143374025821686, "rewards/total_composite/std": 0.11004206538200378, "sampling/importance_sampling_ratio/max": 1.671715497970581, "sampling/importance_sampling_ratio/mean": 0.9923664331436157, "sampling/importance_sampling_ratio/min": 0.13702690601348877, "sampling/sampling_logp_difference/max": 1.9875779151916504, "sampling/sampling_logp_difference/mean": 0.03102065436542034, "step": 802 }, { "clip_ratio/high_max": 0.026509752846322954, "clip_ratio/high_mean": 0.026509752846322954, "clip_ratio/low_mean": 0.005281690042465925, "clip_ratio/low_min": 0.005281690042465925, "clip_ratio/region_mean": 0.03179144288878888, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 70.75, "completions/mean_terminated_length": 70.75, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.1510819010436535, "epoch": 0.03225288187331807, "frac_reward_zero_std": 0.0, "grad_norm": 4.46676778793335, "learning_rate": 7.56969696969697e-06, "loss": 0.0049, "num_tokens": 1775325.0, "reward": 0.9594069123268127, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9594069123268127, "reward_meter_std": 0.05366376414895058, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05366375669836998, "reward_total_composite_mean": 0.9594069123268127, "reward_total_composite_std": 0.05366376414895058, "reward_total_mean": 0.9594069123268127, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9594069123268127, "rewards/meter/std": 0.05366376414895058, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9594069123268127, "rewards/total_composite/std": 0.05366376414895058, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9985708594322205, "sampling/importance_sampling_ratio/min": 0.14877213537693024, "sampling/sampling_logp_difference/max": 1.9053394794464111, "sampling/sampling_logp_difference/mean": 0.03665493428707123, "step": 803 }, { "clip_ratio/high_max": 0.003243225917685777, "clip_ratio/high_mean": 0.003243225917685777, "clip_ratio/low_mean": 0.0004926113178953528, "clip_ratio/low_min": 0.0004926113178953528, "clip_ratio/region_mean": 0.00373583723558113, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 508.0, "completions/mean_length": 505.25, "completions/mean_terminated_length": 503.0, "completions/min_length": 493.0, "completions/min_terminated_length": 493.0, "entropy": 0.014583299867808819, "epoch": 0.03229304735510302, "frac_reward_zero_std": 0.0, "grad_norm": 0.5559709072113037, "learning_rate": 7.566666666666667e-06, "loss": 0.0506, "num_tokens": 1780007.0, "reward": 0.38479870557785034, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.0235702246427536, "reward_meter_mean": 0.9569629430770874, "reward_meter_std": 0.102637879550457, "reward_repeat_penalty_mean": 0.4376780688762665, "reward_repeat_penalty_std": 0.22492921352386475, "reward_std": 0.20677603781223297, "reward_total_composite_mean": 0.38479870557785034, "reward_total_composite_std": 0.20677605271339417, "reward_total_mean": 0.38479870557785034, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.0235702246427536, "rewards/meter/mean": 0.9569629430770874, "rewards/meter/std": 0.102637879550457, "rewards/repeat_penalty/mean": 0.4376780688762665, "rewards/repeat_penalty/std": 0.22492921352386475, "rewards/total_composite/mean": 0.38479870557785034, "rewards/total_composite/std": 0.20677605271339417, "sampling/importance_sampling_ratio/max": 1.9659916162490845, "sampling/importance_sampling_ratio/mean": 1.0004603862762451, "sampling/importance_sampling_ratio/min": 0.06372835487127304, "sampling/sampling_logp_difference/max": 2.7531256675720215, "sampling/sampling_logp_difference/mean": 0.006570629775524139, "step": 804 }, { "clip_ratio/high_max": 0.00932835799176246, "clip_ratio/high_mean": 0.00932835799176246, "clip_ratio/low_mean": 0.011085874866694212, "clip_ratio/low_min": 0.011085874866694212, "clip_ratio/region_mean": 0.02041423285845667, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.25, "completions/mean_terminated_length": 67.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.05657489877194166, "epoch": 0.032333212836887976, "frac_reward_zero_std": 0.0, "grad_norm": 4.009762287139893, "learning_rate": 7.563636363636364e-06, "loss": 0.0024, "num_tokens": 1781817.0, "reward": 0.7574520111083984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7574520111083984, "reward_meter_std": 0.03009505569934845, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0300950538367033, "reward_total_composite_mean": 0.7574520111083984, "reward_total_composite_std": 0.03009505569934845, "reward_total_mean": 0.7574520111083984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7574520111083984, "rewards/meter/std": 0.03009505569934845, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7574520111083984, "rewards/total_composite/std": 0.03009505569934845, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9979568719863892, "sampling/importance_sampling_ratio/min": 0.29304373264312744, "sampling/sampling_logp_difference/max": 1.227433443069458, "sampling/sampling_logp_difference/mean": 0.020206134766340256, "step": 805 }, { "clip_ratio/high_max": 0.01062974869273603, "clip_ratio/high_mean": 0.01062974869273603, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/region_mean": 0.01494009350426495, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 58.125, "completions/mean_terminated_length": 58.125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.11172830406576395, "epoch": 0.03237337831867293, "frac_reward_zero_std": 0.0, "grad_norm": 3.6810977458953857, "learning_rate": 7.560606060606062e-06, "loss": 0.0021, "num_tokens": 1783594.0, "reward": 0.8252000212669373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.990125298500061, "reward_meter_std": 0.0014512698398903012, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_std": 0.17693018913269043, "reward_total_composite_mean": 0.8252000212669373, "reward_total_composite_std": 0.17693018913269043, "reward_total_mean": 0.8252000212669373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.990125298500061, "rewards/meter/std": 0.0014512698398903012, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.8252000212669373, "rewards/total_composite/std": 0.17693018913269043, "sampling/importance_sampling_ratio/max": 1.3977869749069214, "sampling/importance_sampling_ratio/mean": 0.9979320168495178, "sampling/importance_sampling_ratio/min": 0.30750730633735657, "sampling/sampling_logp_difference/max": 1.1792564392089844, "sampling/sampling_logp_difference/mean": 0.023440338671207428, "step": 806 }, { "clip_ratio/high_max": 0.008098726975731552, "clip_ratio/high_mean": 0.008098726975731552, "clip_ratio/low_mean": 0.00854225957300514, "clip_ratio/low_min": 0.00854225957300514, "clip_ratio/region_mean": 0.01664098654873669, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 75.875, "completions/mean_terminated_length": 75.875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.12689008563756943, "epoch": 0.032413543800457884, "frac_reward_zero_std": 0.0, "grad_norm": 5.640214920043945, "learning_rate": 7.557575757575758e-06, "loss": -0.0084, "num_tokens": 1785577.0, "reward": 0.7303093671798706, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.7918260097503662, "reward_meter_std": 0.3541868031024933, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35869717597961426, "reward_total_composite_mean": 0.7303093671798706, "reward_total_composite_std": 0.35869720578193665, "reward_total_mean": 0.7303093671798706, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.7918260097503662, "rewards/meter/std": 0.3541868031024933, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7303093671798706, "rewards/total_composite/std": 0.35869720578193665, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0031059980392456, "sampling/importance_sampling_ratio/min": 0.4486599266529083, "sampling/sampling_logp_difference/max": 1.1890771389007568, "sampling/sampling_logp_difference/mean": 0.02571277692914009, "step": 807 }, { "clip_ratio/high_max": 0.016447368543595076, "clip_ratio/high_mean": 0.016447368543595076, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.016447368543595076, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 38.0, "completions/mean_terminated_length": 38.0, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.09128655772656202, "epoch": 0.03245370928224284, "frac_reward_zero_std": 0.0, "grad_norm": 14.179214477539062, "learning_rate": 7.5545454545454555e-06, "loss": -0.0018, "num_tokens": 1787137.0, "reward": 0.8494601845741272, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8494601845741272, "reward_meter_std": 0.3245563507080078, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3245563209056854, "reward_total_composite_mean": 0.8494601845741272, "reward_total_composite_std": 0.3245563507080078, "reward_total_mean": 0.8494601845741272, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8494601845741272, "rewards/meter/std": 0.3245563507080078, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8494601845741272, "rewards/total_composite/std": 0.3245563507080078, "sampling/importance_sampling_ratio/max": 1.5366783142089844, "sampling/importance_sampling_ratio/mean": 0.9977914690971375, "sampling/importance_sampling_ratio/min": 0.11267625540494919, "sampling/sampling_logp_difference/max": 2.183236598968506, "sampling/sampling_logp_difference/mean": 0.026692412793636322, "step": 808 }, { "clip_ratio/high_max": 0.003968499368056655, "clip_ratio/high_mean": 0.003968499368056655, "clip_ratio/low_mean": 0.005849295761436224, "clip_ratio/low_min": 0.005849295761436224, "clip_ratio/region_mean": 0.009817795129492879, "completions/clipped_ratio": 0.0, "completions/max_length": 230.0, "completions/max_terminated_length": 230.0, "completions/mean_length": 218.25, "completions/mean_terminated_length": 218.25, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.03606141824275255, "epoch": 0.03249387476402779, "frac_reward_zero_std": 0.0, "grad_norm": 1.1023591756820679, "learning_rate": 7.551515151515152e-06, "loss": -0.0031, "num_tokens": 1790539.0, "reward": 0.5638895034790039, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9922986030578613, "reward_meter_std": 0.004149852320551872, "reward_repeat_penalty_mean": 0.5681818723678589, "reward_repeat_penalty_std": 0.1735115498304367, "reward_std": 0.17219732701778412, "reward_total_composite_mean": 0.5638895034790039, "reward_total_composite_std": 0.17219732701778412, "reward_total_mean": 0.5638895034790039, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9922986030578613, "rewards/meter/std": 0.004149852320551872, "rewards/repeat_penalty/mean": 0.5681818723678589, "rewards/repeat_penalty/std": 0.1735115498304367, "rewards/total_composite/mean": 0.5638895034790039, "rewards/total_composite/std": 0.17219732701778412, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0026495456695557, "sampling/importance_sampling_ratio/min": 0.20739488303661346, "sampling/sampling_logp_difference/max": 1.5731306076049805, "sampling/sampling_logp_difference/mean": 0.011151120997965336, "step": 809 }, { "clip_ratio/high_max": 0.03143186215311289, "clip_ratio/high_mean": 0.03143186215311289, "clip_ratio/low_mean": 0.005306840990670025, "clip_ratio/low_min": 0.005306840990670025, "clip_ratio/region_mean": 0.036738703143782914, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.15044390503317118, "epoch": 0.032534040245812745, "frac_reward_zero_std": 0.0, "grad_norm": 5.80548095703125, "learning_rate": 7.548484848484849e-06, "loss": 0.0017, "num_tokens": 1792366.0, "reward": 0.9332020878791809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9742876887321472, "reward_meter_std": 0.029914017766714096, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11529984325170517, "reward_total_composite_mean": 0.9332020878791809, "reward_total_composite_std": 0.11529982835054398, "reward_total_mean": 0.9332020878791809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9742876887321472, "rewards/meter/std": 0.029914017766714096, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9332020878791809, "rewards/total_composite/std": 0.11529982835054398, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0060287714004517, "sampling/importance_sampling_ratio/min": 0.14418534934520721, "sampling/sampling_logp_difference/max": 1.9366556406021118, "sampling/sampling_logp_difference/mean": 0.0373886339366436, "step": 810 }, { "clip_ratio/high_max": 0.011296228156425059, "clip_ratio/high_mean": 0.011296228156425059, "clip_ratio/low_mean": 0.006675369921140373, "clip_ratio/low_min": 0.006675369921140373, "clip_ratio/region_mean": 0.01797159807756543, "completions/clipped_ratio": 0.0, "completions/max_length": 229.0, "completions/max_terminated_length": 229.0, "completions/mean_length": 203.75, "completions/mean_terminated_length": 203.75, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "entropy": 0.1404771413654089, "epoch": 0.0325742057275977, "frac_reward_zero_std": 0.0, "grad_norm": 2.172598123550415, "learning_rate": 7.545454545454546e-06, "loss": 0.0192, "num_tokens": 1795876.0, "reward": 0.3340369462966919, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.710443377494812, "reward_meter_std": 0.2666179835796356, "reward_repeat_penalty_mean": 0.4431818127632141, "reward_repeat_penalty_std": 0.22498852014541626, "reward_std": 0.2214316576719284, "reward_total_composite_mean": 0.3340369462966919, "reward_total_composite_std": 0.2214316725730896, "reward_total_mean": 0.3340369462966919, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.710443377494812, "rewards/meter/std": 0.2666179835796356, "rewards/repeat_penalty/mean": 0.4431818127632141, "rewards/repeat_penalty/std": 0.22498852014541626, "rewards/total_composite/mean": 0.3340369462966919, "rewards/total_composite/std": 0.2214316725730896, "sampling/importance_sampling_ratio/max": 1.9448559284210205, "sampling/importance_sampling_ratio/mean": 1.0012760162353516, "sampling/importance_sampling_ratio/min": 0.012466873973608017, "sampling/sampling_logp_difference/max": 4.384680271148682, "sampling/sampling_logp_difference/mean": 0.021875398233532906, "step": 811 }, { "clip_ratio/high_max": 0.005769230774603784, "clip_ratio/high_mean": 0.005769230774603784, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005769230774603784, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.25, "completions/mean_terminated_length": 65.25, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.036765412194654346, "epoch": 0.03261437120938265, "frac_reward_zero_std": 0.0, "grad_norm": 8.286386489868164, "learning_rate": 7.542424242424244e-06, "loss": 0.0139, "num_tokens": 1797566.0, "reward": 0.9563552141189575, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978810548782349, "reward_meter_std": 0.0005103643052279949, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11796264350414276, "reward_total_composite_mean": 0.9563552141189575, "reward_total_composite_std": 0.11796265840530396, "reward_total_mean": 0.9563552141189575, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978810548782349, "rewards/meter/std": 0.0005103643052279949, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9563552141189575, "rewards/total_composite/std": 0.11796265840530396, "sampling/importance_sampling_ratio/max": 1.3499990701675415, "sampling/importance_sampling_ratio/mean": 1.002901315689087, "sampling/importance_sampling_ratio/min": 0.5553844571113586, "sampling/sampling_logp_difference/max": 0.5880947113037109, "sampling/sampling_logp_difference/mean": 0.007206229493021965, "step": 812 }, { "clip_ratio/high_max": 0.0019113150192424655, "clip_ratio/high_mean": 0.0019113150192424655, "clip_ratio/low_mean": 0.005218350415816531, "clip_ratio/low_min": 0.005218350415816531, "clip_ratio/region_mean": 0.007129665435058996, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 332.375, "completions/mean_terminated_length": 332.375, "completions/min_length": 302.0, "completions/min_terminated_length": 302.0, "entropy": 0.02348946128040552, "epoch": 0.03265453669116761, "frac_reward_zero_std": 0.0, "grad_norm": 1.0310794115066528, "learning_rate": 7.53939393939394e-06, "loss": 0.0079, "num_tokens": 1801761.0, "reward": 0.517345666885376, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9938352704048157, "reward_meter_std": 0.008550022728741169, "reward_repeat_penalty_mean": 0.5850183963775635, "reward_repeat_penalty_std": 0.009098809212446213, "reward_std": 0.012124458327889442, "reward_total_composite_mean": 0.517345666885376, "reward_total_composite_std": 0.012124452739953995, "reward_total_mean": 0.517345666885376, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9938352704048157, "rewards/meter/std": 0.008550022728741169, "rewards/repeat_penalty/mean": 0.5850183963775635, "rewards/repeat_penalty/std": 0.009098809212446213, "rewards/total_composite/mean": 0.517345666885376, "rewards/total_composite/std": 0.012124452739953995, "sampling/importance_sampling_ratio/max": 1.6740483045578003, "sampling/importance_sampling_ratio/mean": 1.0009092092514038, "sampling/importance_sampling_ratio/min": 0.47178637981414795, "sampling/sampling_logp_difference/max": 0.7512289881706238, "sampling/sampling_logp_difference/mean": 0.0052232746966183186, "step": 813 }, { "clip_ratio/high_max": 0.00237341778120026, "clip_ratio/high_mean": 0.00237341778120026, "clip_ratio/low_mean": 0.00237341778120026, "clip_ratio/low_min": 0.00237341778120026, "clip_ratio/region_mean": 0.00474683556240052, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 159.375, "completions/mean_terminated_length": 159.375, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.03128327080048621, "epoch": 0.03269470217295257, "frac_reward_zero_std": 0.0, "grad_norm": 13.524036407470703, "learning_rate": 7.536363636363637e-06, "loss": -0.001, "num_tokens": 1804516.0, "reward": 0.6653216481208801, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979825019836426, "reward_meter_std": 0.0006693408940918744, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00044622053974308074, "reward_total_composite_mean": 0.6653216481208801, "reward_total_composite_std": 0.00044622019049711525, "reward_total_mean": 0.6653216481208801, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979825019836426, "rewards/meter/std": 0.0006693408940918744, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6653216481208801, "rewards/total_composite/std": 0.00044622019049711525, "sampling/importance_sampling_ratio/max": 1.493585467338562, "sampling/importance_sampling_ratio/mean": 0.9996488094329834, "sampling/importance_sampling_ratio/min": 0.10725226998329163, "sampling/sampling_logp_difference/max": 2.232571601867676, "sampling/sampling_logp_difference/mean": 0.005735776387155056, "step": 814 }, { "clip_ratio/high_max": 0.016365812392905354, "clip_ratio/high_mean": 0.016365812392905354, "clip_ratio/low_mean": 0.011092530796304345, "clip_ratio/low_min": 0.011092530796304345, "clip_ratio/region_mean": 0.0274583431892097, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 62.125, "completions/mean_terminated_length": 62.125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.10703426506370306, "epoch": 0.03273486765473752, "frac_reward_zero_std": 0.0, "grad_norm": 8.952130317687988, "learning_rate": 7.533333333333334e-06, "loss": 0.0458, "num_tokens": 1806317.0, "reward": 0.8894620537757874, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9308251142501831, "reward_meter_std": 0.15261007845401764, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.1764250546693802, "reward_total_composite_mean": 0.8894620537757874, "reward_total_composite_std": 0.17642506957054138, "reward_total_mean": 0.8894620537757874, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9308251142501831, "rewards/meter/std": 0.15261007845401764, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8894620537757874, "rewards/total_composite/std": 0.17642506957054138, "sampling/importance_sampling_ratio/max": 1.5671416521072388, "sampling/importance_sampling_ratio/mean": 0.9998781085014343, "sampling/importance_sampling_ratio/min": 0.2406802922487259, "sampling/sampling_logp_difference/max": 1.424285888671875, "sampling/sampling_logp_difference/mean": 0.024382825940847397, "step": 815 }, { "clip_ratio/high_max": 0.028449730249121785, "clip_ratio/high_mean": 0.028449730249121785, "clip_ratio/low_mean": 0.023026316426694393, "clip_ratio/low_min": 0.023026316426694393, "clip_ratio/region_mean": 0.05147604667581618, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 39.0, "completions/mean_terminated_length": 39.0, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.16819044947624207, "epoch": 0.032775033136522476, "frac_reward_zero_std": 0.0, "grad_norm": 8.190361022949219, "learning_rate": 7.530303030303031e-06, "loss": -0.0066, "num_tokens": 1807821.0, "reward": 0.7446458339691162, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7446458339691162, "reward_meter_std": 0.30041757225990295, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.30041757225990295, "reward_total_composite_mean": 0.7446458339691162, "reward_total_composite_std": 0.30041757225990295, "reward_total_mean": 0.7446458339691162, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7446458339691162, "rewards/meter/std": 0.30041757225990295, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7446458339691162, "rewards/total_composite/std": 0.30041757225990295, "sampling/importance_sampling_ratio/max": 1.8297884464263916, "sampling/importance_sampling_ratio/mean": 0.9998385310173035, "sampling/importance_sampling_ratio/min": 0.12064976990222931, "sampling/sampling_logp_difference/max": 2.114863395690918, "sampling/sampling_logp_difference/mean": 0.059971921145915985, "step": 816 }, { "clip_ratio/high_max": 0.013432800536975265, "clip_ratio/high_mean": 0.013432800536975265, "clip_ratio/low_mean": 0.01120058260858059, "clip_ratio/low_min": 0.01120058260858059, "clip_ratio/region_mean": 0.024633383145555854, "completions/clipped_ratio": 0.0, "completions/max_length": 115.0, "completions/max_terminated_length": 115.0, "completions/mean_length": 107.125, "completions/mean_terminated_length": 107.125, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.05588802928104997, "epoch": 0.03281519861830743, "frac_reward_zero_std": 0.0, "grad_norm": 3.7302210330963135, "learning_rate": 7.5272727272727274e-06, "loss": 0.0413, "num_tokens": 1810070.0, "reward": 0.2755756676197052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3294169008731842, "reward_meter_std": 0.28731057047843933, "reward_repeat_penalty_mean": 0.7750000357627869, "reward_repeat_penalty_std": 0.12817399203777313, "reward_std": 0.2410556823015213, "reward_total_composite_mean": 0.2755756676197052, "reward_total_composite_std": 0.2410556823015213, "reward_total_mean": 0.2755756676197052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3294169008731842, "rewards/meter/std": 0.28731057047843933, "rewards/repeat_penalty/mean": 0.7750000357627869, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.2755756676197052, "rewards/total_composite/std": 0.2410556823015213, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0044699907302856, "sampling/importance_sampling_ratio/min": 0.2369401901960373, "sampling/sampling_logp_difference/max": 1.4399476051330566, "sampling/sampling_logp_difference/mean": 0.01802036352455616, "step": 817 }, { "clip_ratio/high_max": 0.006020042230375111, "clip_ratio/high_mean": 0.006020042230375111, "clip_ratio/low_mean": 0.003796361561398953, "clip_ratio/low_min": 0.003796361561398953, "clip_ratio/region_mean": 0.009816403791774064, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 357.75, "completions/mean_terminated_length": 357.75, "completions/min_length": 346.0, "completions/min_terminated_length": 346.0, "entropy": 0.03675817488692701, "epoch": 0.032855364100092384, "frac_reward_zero_std": 0.0, "grad_norm": 1.0928229093551636, "learning_rate": 7.524242424242425e-06, "loss": 0.0108, "num_tokens": 1814452.0, "reward": 0.5283541083335876, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.05750546231865883, "reward_meter_mean": 0.9965507984161377, "reward_meter_std": 0.0037117046304047108, "reward_repeat_penalty_mean": 0.5523655414581299, "reward_repeat_penalty_std": 0.10810358822345734, "reward_std": 0.11482104659080505, "reward_total_composite_mean": 0.5283541083335876, "reward_total_composite_std": 0.11482103914022446, "reward_total_mean": 0.5283541083335876, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.05750546231865883, "rewards/meter/mean": 0.9965507984161377, "rewards/meter/std": 0.0037117046304047108, "rewards/repeat_penalty/mean": 0.5523655414581299, "rewards/repeat_penalty/std": 0.10810358822345734, "rewards/total_composite/mean": 0.5283541083335876, "rewards/total_composite/std": 0.11482103914022446, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0001471042633057, "sampling/importance_sampling_ratio/min": 0.3455010652542114, "sampling/sampling_logp_difference/max": 1.062759518623352, "sampling/sampling_logp_difference/mean": 0.00884958729147911, "step": 818 }, { "clip_ratio/high_max": 0.022682114504277706, "clip_ratio/high_mean": 0.022682114504277706, "clip_ratio/low_mean": 0.013770792167633772, "clip_ratio/low_min": 0.013770792167633772, "clip_ratio/region_mean": 0.03645290667191148, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 61.625, "completions/mean_terminated_length": 61.625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.10824673250317574, "epoch": 0.03289552958187734, "frac_reward_zero_std": 0.0, "grad_norm": 6.570438861846924, "learning_rate": 7.521212121212121e-06, "loss": 0.0279, "num_tokens": 1816193.0, "reward": 0.9746619462966919, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9746619462966919, "reward_meter_std": 0.020627282559871674, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.020627308636903763, "reward_total_composite_mean": 0.9746619462966919, "reward_total_composite_std": 0.020627282559871674, "reward_total_mean": 0.9746619462966919, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9746619462966919, "rewards/meter/std": 0.020627282559871674, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9746619462966919, "rewards/total_composite/std": 0.020627282559871674, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9997745156288147, "sampling/importance_sampling_ratio/min": 0.26106762886047363, "sampling/sampling_logp_difference/max": 1.3429758548736572, "sampling/sampling_logp_difference/mean": 0.02879003807902336, "step": 819 }, { "clip_ratio/high_max": 0.008506174897775054, "clip_ratio/high_mean": 0.008506174897775054, "clip_ratio/low_mean": 0.007311595429200679, "clip_ratio/low_min": 0.007311595429200679, "clip_ratio/region_mean": 0.015817770326975733, "completions/clipped_ratio": 0.0, "completions/max_length": 156.0, "completions/max_terminated_length": 156.0, "completions/mean_length": 149.375, "completions/mean_terminated_length": 149.375, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.06724543264135718, "epoch": 0.03293569506366229, "frac_reward_zero_std": 0.0, "grad_norm": 2.3457443714141846, "learning_rate": 7.518181818181819e-06, "loss": 0.0185, "num_tokens": 1818812.0, "reward": 0.4300089478492737, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7805307507514954, "reward_meter_std": 0.1253480166196823, "reward_repeat_penalty_mean": 0.5714285373687744, "reward_repeat_penalty_std": 0.17074695229530334, "reward_std": 0.0887664407491684, "reward_total_composite_mean": 0.4300089478492737, "reward_total_composite_std": 0.0887664407491684, "reward_total_mean": 0.4300089478492737, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7805307507514954, "rewards/meter/std": 0.1253480166196823, "rewards/repeat_penalty/mean": 0.5714285373687744, "rewards/repeat_penalty/std": 0.17074695229530334, "rewards/total_composite/mean": 0.4300089478492737, "rewards/total_composite/std": 0.0887664407491684, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0010299682617188, "sampling/importance_sampling_ratio/min": 0.3003794550895691, "sampling/sampling_logp_difference/max": 1.2027087211608887, "sampling/sampling_logp_difference/mean": 0.015637392178177834, "step": 820 }, { "clip_ratio/high_max": 0.01558885129634291, "clip_ratio/high_mean": 0.01558885129634291, "clip_ratio/low_mean": 0.007159017724916339, "clip_ratio/low_min": 0.007159017724916339, "clip_ratio/region_mean": 0.022747869021259248, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 190.25, "completions/mean_terminated_length": 83.0, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.07134009897708893, "epoch": 0.032975860545447246, "frac_reward_zero_std": 0.0, "grad_norm": 0.9305291175842285, "learning_rate": 7.515151515151516e-06, "loss": -0.1471, "num_tokens": 1820662.0, "reward": 0.5668963193893433, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_meter_mean": 0.5668963193893433, "reward_meter_std": 0.39650967717170715, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.39650964736938477, "reward_total_composite_mean": 0.5668963193893433, "reward_total_composite_std": 0.39650967717170715, "reward_total_mean": 0.5668963193893433, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/meter/mean": 0.5668963193893433, "rewards/meter/std": 0.39650967717170715, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5668963193893433, "rewards/total_composite/std": 0.39650967717170715, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0027402639389038, "sampling/importance_sampling_ratio/min": 0.6215139627456665, "sampling/sampling_logp_difference/max": 0.7466448545455933, "sampling/sampling_logp_difference/mean": 0.019788384437561035, "step": 821 }, { "clip_ratio/high_max": 0.009740259731188416, "clip_ratio/high_mean": 0.009740259731188416, "clip_ratio/low_mean": 0.006382113788276911, "clip_ratio/low_min": 0.006382113788276911, "clip_ratio/region_mean": 0.016122373519465327, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 77.875, "completions/mean_terminated_length": 77.875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.09962678607553244, "epoch": 0.0330160260272322, "frac_reward_zero_std": 0.0, "grad_norm": 3.704124689102173, "learning_rate": 7.512121212121213e-06, "loss": 0.0105, "num_tokens": 1822669.0, "reward": 0.8737984299659729, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8737984299659729, "reward_meter_std": 0.14763900637626648, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.14763899147510529, "reward_total_composite_mean": 0.8737984299659729, "reward_total_composite_std": 0.14763900637626648, "reward_total_mean": 0.8737984299659729, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8737984299659729, "rewards/meter/std": 0.14763900637626648, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8737984299659729, "rewards/total_composite/std": 0.14763900637626648, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0058021545410156, "sampling/importance_sampling_ratio/min": 0.28069284558296204, "sampling/sampling_logp_difference/max": 1.2704942226409912, "sampling/sampling_logp_difference/mean": 0.018405567854642868, "step": 822 }, { "clip_ratio/high_max": 0.0065078792395070195, "clip_ratio/high_mean": 0.0065078792395070195, "clip_ratio/low_mean": 0.002257478976389393, "clip_ratio/low_min": 0.002257478976389393, "clip_ratio/region_mean": 0.008765358215896413, "completions/clipped_ratio": 0.0, "completions/max_length": 297.0, "completions/max_terminated_length": 297.0, "completions/mean_length": 284.375, "completions/mean_terminated_length": 284.375, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 0.03474967950023711, "epoch": 0.033056191509017153, "frac_reward_zero_std": 0.0, "grad_norm": 2.836785078048706, "learning_rate": 7.509090909090909e-06, "loss": 0.0055, "num_tokens": 1826360.0, "reward": 0.5862284898757935, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9821428656578064, "reward_count_adherence_std": 0.05050762742757797, "reward_meter_mean": 0.9983997344970703, "reward_meter_std": 0.0007759786094538867, "reward_repeat_penalty_mean": 0.5952796936035156, "reward_repeat_penalty_std": 0.08133196830749512, "reward_std": 0.09859946370124817, "reward_total_composite_mean": 0.5862284898757935, "reward_total_composite_std": 0.09859947860240936, "reward_total_mean": 0.5862284898757935, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9821428656578064, "rewards/count_adherence/std": 0.05050762742757797, "rewards/meter/mean": 0.9983997344970703, "rewards/meter/std": 0.0007759786094538867, "rewards/repeat_penalty/mean": 0.5952796936035156, "rewards/repeat_penalty/std": 0.08133196830749512, "rewards/total_composite/mean": 0.5862284898757935, "rewards/total_composite/std": 0.09859947860240936, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0005059242248535, "sampling/importance_sampling_ratio/min": 0.07232620567083359, "sampling/sampling_logp_difference/max": 2.6265687942504883, "sampling/sampling_logp_difference/mean": 0.015353661961853504, "step": 823 }, { "clip_ratio/high_max": 0.0070436508394777775, "clip_ratio/high_mean": 0.0070436508394777775, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/region_mean": 0.008779761963523924, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.0, "completions/mean_terminated_length": 71.0, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.04505776287987828, "epoch": 0.03309635699080211, "frac_reward_zero_std": 0.0, "grad_norm": 4.631432056427002, "learning_rate": 7.5060606060606065e-06, "loss": 0.0168, "num_tokens": 1828232.0, "reward": 0.7449524998664856, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7449524998664856, "reward_meter_std": 0.04608132317662239, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04608132690191269, "reward_total_composite_mean": 0.7449524998664856, "reward_total_composite_std": 0.04608132317662239, "reward_total_mean": 0.7449524998664856, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7449524998664856, "rewards/meter/std": 0.04608132317662239, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7449524998664856, "rewards/total_composite/std": 0.04608132317662239, "sampling/importance_sampling_ratio/max": 1.426641583442688, "sampling/importance_sampling_ratio/mean": 0.9986546039581299, "sampling/importance_sampling_ratio/min": 0.1190476045012474, "sampling/sampling_logp_difference/max": 2.1282317638397217, "sampling/sampling_logp_difference/mean": 0.01667422614991665, "step": 824 }, { "clip_ratio/high_max": 0.028171515092253685, "clip_ratio/high_mean": 0.028171515092253685, "clip_ratio/low_mean": 0.024154700804501772, "clip_ratio/low_min": 0.024154700804501772, "clip_ratio/region_mean": 0.05232621589675546, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 68.375, "completions/mean_terminated_length": 68.375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.10288840066641569, "epoch": 0.03313652247258706, "frac_reward_zero_std": 0.0, "grad_norm": 4.750690937042236, "learning_rate": 7.503030303030303e-06, "loss": 0.0481, "num_tokens": 1830123.0, "reward": 0.4837471544742584, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4837471544742584, "reward_meter_std": 0.22915734350681305, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22915734350681305, "reward_total_composite_mean": 0.4837471544742584, "reward_total_composite_std": 0.22915734350681305, "reward_total_mean": 0.4837471544742584, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4837471544742584, "rewards/meter/std": 0.22915734350681305, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4837471544742584, "rewards/total_composite/std": 0.22915734350681305, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9999799728393555, "sampling/importance_sampling_ratio/min": 0.15487320721149445, "sampling/sampling_logp_difference/max": 1.8651485443115234, "sampling/sampling_logp_difference/mean": 0.031915146857500076, "step": 825 }, { "clip_ratio/high_max": 0.005249909590929747, "clip_ratio/high_mean": 0.005249909590929747, "clip_ratio/low_mean": 0.0033315176842734218, "clip_ratio/low_min": 0.0033315176842734218, "clip_ratio/region_mean": 0.008581427275203168, "completions/clipped_ratio": 0.0, "completions/max_length": 421.0, "completions/max_terminated_length": 421.0, "completions/mean_length": 400.125, "completions/mean_terminated_length": 400.125, "completions/min_length": 375.0, "completions/min_terminated_length": 375.0, "entropy": 0.0334128841641359, "epoch": 0.033176687954372015, "frac_reward_zero_std": 0.0, "grad_norm": 1.1773563623428345, "learning_rate": 7.500000000000001e-06, "loss": 0.0159, "num_tokens": 1835244.0, "reward": 0.3971686363220215, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9545454978942871, "reward_count_adherence_std": 0.0485929399728775, "reward_meter_mean": 0.7846977710723877, "reward_meter_std": 0.39648592472076416, "reward_repeat_penalty_mean": 0.5507364273071289, "reward_repeat_penalty_std": 0.050370436161756516, "reward_std": 0.1935662031173706, "reward_total_composite_mean": 0.3971686363220215, "reward_total_composite_std": 0.1935662180185318, "reward_total_mean": 0.3971686363220215, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9545454978942871, "rewards/count_adherence/std": 0.0485929399728775, "rewards/meter/mean": 0.7846977710723877, "rewards/meter/std": 0.39648592472076416, "rewards/repeat_penalty/mean": 0.5507364273071289, "rewards/repeat_penalty/std": 0.050370436161756516, "rewards/total_composite/mean": 0.3971686363220215, "rewards/total_composite/std": 0.1935662180185318, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0009510517120361, "sampling/importance_sampling_ratio/min": 0.25283992290496826, "sampling/sampling_logp_difference/max": 1.5345345735549927, "sampling/sampling_logp_difference/mean": 0.007981406524777412, "step": 826 }, { "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/low_mean": 0.009191176504828036, "clip_ratio/low_min": 0.009191176504828036, "clip_ratio/region_mean": 0.014625959214754403, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.25, "completions/mean_terminated_length": 68.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.05680654477328062, "epoch": 0.03321685343615697, "frac_reward_zero_std": 0.0, "grad_norm": 3.7937092781066895, "learning_rate": 7.496969696969698e-06, "loss": -0.0012, "num_tokens": 1837278.0, "reward": 0.15423895418643951, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.15423895418643951, "reward_meter_std": 0.07883358746767044, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.07883358746767044, "reward_total_composite_mean": 0.15423895418643951, "reward_total_composite_std": 0.07883358746767044, "reward_total_mean": 0.15423895418643951, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.15423895418643951, "rewards/meter/std": 0.07883358746767044, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.15423895418643951, "rewards/total_composite/std": 0.07883358746767044, "sampling/importance_sampling_ratio/max": 1.73274564743042, "sampling/importance_sampling_ratio/mean": 0.9993041157722473, "sampling/importance_sampling_ratio/min": 0.5242664217948914, "sampling/sampling_logp_difference/max": 0.6457552909851074, "sampling/sampling_logp_difference/mean": 0.016422174870967865, "step": 827 }, { "clip_ratio/high_max": 0.01206992007791996, "clip_ratio/high_mean": 0.01206992007791996, "clip_ratio/low_mean": 0.0038470644503831863, "clip_ratio/low_min": 0.0038470644503831863, "clip_ratio/region_mean": 0.015916984528303146, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 63.5, "completions/mean_terminated_length": 63.5, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.11623660661280155, "epoch": 0.03325701891794192, "frac_reward_zero_std": 0.0, "grad_norm": 6.0185723304748535, "learning_rate": 7.493939393939395e-06, "loss": 0.0145, "num_tokens": 1839058.0, "reward": 0.9804123640060425, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9804123640060425, "reward_meter_std": 0.014910156838595867, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014910157769918442, "reward_total_composite_mean": 0.9804123640060425, "reward_total_composite_std": 0.014910156838595867, "reward_total_mean": 0.9804123640060425, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9804123640060425, "rewards/meter/std": 0.014910156838595867, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9804123640060425, "rewards/total_composite/std": 0.014910156838595867, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0006060600280762, "sampling/importance_sampling_ratio/min": 0.25273311138153076, "sampling/sampling_logp_difference/max": 1.3754212856292725, "sampling/sampling_logp_difference/mean": 0.028718305751681328, "step": 828 }, { "clip_ratio/high_max": 0.016328177880495787, "clip_ratio/high_mean": 0.016328177880495787, "clip_ratio/low_mean": 0.002016129030380398, "clip_ratio/low_min": 0.002016129030380398, "clip_ratio/region_mean": 0.018344306910876185, "completions/clipped_ratio": 0.0, "completions/max_length": 186.0, "completions/max_terminated_length": 186.0, "completions/mean_length": 173.375, "completions/mean_terminated_length": 173.375, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.05448292032815516, "epoch": 0.03329718439972688, "frac_reward_zero_std": 0.0, "grad_norm": 1.7107356786727905, "learning_rate": 7.490909090909092e-06, "loss": 0.0432, "num_tokens": 1841757.0, "reward": 0.6229597926139832, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961064457893372, "reward_meter_std": 0.0029631657525897026, "reward_repeat_penalty_mean": 0.625, "reward_repeat_penalty_std": 0.16967642307281494, "reward_std": 0.17044374346733093, "reward_total_composite_mean": 0.6229597926139832, "reward_total_composite_std": 0.17044374346733093, "reward_total_mean": 0.6229597926139832, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961064457893372, "rewards/meter/std": 0.0029631657525897026, "rewards/repeat_penalty/mean": 0.625, "rewards/repeat_penalty/std": 0.16967642307281494, "rewards/total_composite/mean": 0.6229597926139832, "rewards/total_composite/std": 0.17044374346733093, "sampling/importance_sampling_ratio/max": 1.7753796577453613, "sampling/importance_sampling_ratio/mean": 0.9979692101478577, "sampling/importance_sampling_ratio/min": 0.22417764365673065, "sampling/sampling_logp_difference/max": 1.495316505432129, "sampling/sampling_logp_difference/mean": 0.017439143732190132, "step": 829 }, { "clip_ratio/high_max": 0.019965628627687693, "clip_ratio/high_mean": 0.019965628627687693, "clip_ratio/low_mean": 0.007117270142771304, "clip_ratio/low_min": 0.007117270142771304, "clip_ratio/region_mean": 0.027082898770458996, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 74.125, "completions/mean_terminated_length": 74.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.12799056991934776, "epoch": 0.03333734988151183, "frac_reward_zero_std": 0.0, "grad_norm": 12.176047325134277, "learning_rate": 7.487878787878788e-06, "loss": 0.034, "num_tokens": 1843710.0, "reward": 0.9026094079017639, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9026094079017639, "reward_meter_std": 0.14784321188926697, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.14784321188926697, "reward_total_composite_mean": 0.9026094079017639, "reward_total_composite_std": 0.14784321188926697, "reward_total_mean": 0.9026094079017639, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9026094079017639, "rewards/meter/std": 0.14784321188926697, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9026094079017639, "rewards/total_composite/std": 0.14784321188926697, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9942511916160583, "sampling/importance_sampling_ratio/min": 0.0027484570164233446, "sampling/sampling_logp_difference/max": 5.8967156410217285, "sampling/sampling_logp_difference/mean": 0.060764994472265244, "step": 830 }, { "clip_ratio/high_max": 0.01371849060524255, "clip_ratio/high_mean": 0.01371849060524255, "clip_ratio/low_mean": 0.016522707068361342, "clip_ratio/low_min": 0.016522707068361342, "clip_ratio/region_mean": 0.030241197673603892, "completions/clipped_ratio": 0.0, "completions/max_length": 113.0, "completions/max_terminated_length": 113.0, "completions/mean_length": 108.125, "completions/mean_terminated_length": 108.125, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.12897953018546104, "epoch": 0.033377515363296785, "frac_reward_zero_std": 0.0, "grad_norm": 4.106361389160156, "learning_rate": 7.484848484848486e-06, "loss": -0.003, "num_tokens": 1845975.0, "reward": 0.3143823742866516, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3786888122558594, "reward_meter_std": 0.35290807485580444, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.282669335603714, "reward_total_composite_mean": 0.3143823742866516, "reward_total_composite_std": 0.282669335603714, "reward_total_mean": 0.3143823742866516, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3786888122558594, "rewards/meter/std": 0.35290807485580444, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.3143823742866516, "rewards/total_composite/std": 0.282669335603714, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9974236488342285, "sampling/importance_sampling_ratio/min": 0.12783241271972656, "sampling/sampling_logp_difference/max": 2.057035207748413, "sampling/sampling_logp_difference/mean": 0.034197013825178146, "step": 831 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.03341768084508174, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 7.481818181818182e-06, "loss": 0.0, "num_tokens": 1847831.0, "reward": 0.3008078336715698, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8602941036224365, "reward_count_adherence_std": 0.0304440688341856, "reward_meter_mean": 0.6360248923301697, "reward_meter_std": 0.25572946667671204, "reward_repeat_penalty_mean": 0.5477695465087891, "reward_repeat_penalty_std": 0.011733362451195717, "reward_std": 0.12469936162233353, "reward_total_composite_mean": 0.3008078336715698, "reward_total_composite_std": 0.12469936162233353, "reward_total_mean": 0.3008078336715698, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8602941036224365, "rewards/count_adherence/std": 0.0304440688341856, "rewards/meter/mean": 0.6360248923301697, "rewards/meter/std": 0.25572946667671204, "rewards/repeat_penalty/mean": 0.5477695465087891, "rewards/repeat_penalty/std": 0.011733362451195717, "rewards/total_composite/mean": 0.3008078336715698, "rewards/total_composite/std": 0.12469936162233353, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 832 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.04187649488449097, "epoch": 0.03345784632686669, "frac_reward_zero_std": 0.0, "grad_norm": 8.433792114257812, "learning_rate": 7.47878787878788e-06, "loss": 0.0048, "num_tokens": 1849335.0, "reward": 0.9924166798591614, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9924166798591614, "reward_meter_std": 0.00017309709801338613, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00017310312250629067, "reward_total_composite_mean": 0.9924166798591614, "reward_total_composite_std": 0.00017309709801338613, "reward_total_mean": 0.9924166798591614, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9924166798591614, "rewards/meter/std": 0.00017309709801338613, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924166798591614, "rewards/total_composite/std": 0.00017309709801338613, "sampling/importance_sampling_ratio/max": 1.1176583766937256, "sampling/importance_sampling_ratio/mean": 1.0005563497543335, "sampling/importance_sampling_ratio/min": 0.5240997672080994, "sampling/sampling_logp_difference/max": 0.6460732221603394, "sampling/sampling_logp_difference/mean": 0.008245354518294334, "step": 833 }, { "clip_ratio/high_max": 0.009836265817284584, "clip_ratio/high_mean": 0.009836265817284584, "clip_ratio/low_mean": 0.035882155993022025, "clip_ratio/low_min": 0.035882155993022025, "clip_ratio/region_mean": 0.04571842181030661, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.33059297781437635, "epoch": 0.03349801180865165, "frac_reward_zero_std": 0.0, "grad_norm": 7.6343560218811035, "learning_rate": 7.4757575757575765e-06, "loss": 0.0342, "num_tokens": 1851111.0, "reward": 0.40049153566360474, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.40049153566360474, "reward_meter_std": 0.3537753224372864, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3537753224372864, "reward_total_composite_mean": 0.40049153566360474, "reward_total_composite_std": 0.3537753224372864, "reward_total_mean": 0.40049153566360474, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.40049153566360474, "rewards/meter/std": 0.3537753224372864, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.40049153566360474, "rewards/total_composite/std": 0.3537753224372864, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015736818313599, "sampling/importance_sampling_ratio/min": 0.03957962989807129, "sampling/sampling_logp_difference/max": 3.229440689086914, "sampling/sampling_logp_difference/mean": 0.0617922842502594, "step": 834 }, { "clip_ratio/high_max": 0.004172259592451155, "clip_ratio/high_mean": 0.004172259592451155, "clip_ratio/low_mean": 0.0007961783558130264, "clip_ratio/low_min": 0.0007961783558130264, "clip_ratio/region_mean": 0.004968437948264182, "completions/clipped_ratio": 0.0, "completions/max_length": 157.0, "completions/max_terminated_length": 157.0, "completions/mean_length": 150.125, "completions/mean_terminated_length": 150.125, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.03071731375530362, "epoch": 0.0335381772904366, "frac_reward_zero_std": 0.0, "grad_norm": 1.6731441020965576, "learning_rate": 7.472727272727274e-06, "loss": 0.0177, "num_tokens": 1853656.0, "reward": 0.6159030199050903, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9238545894622803, "reward_meter_std": 0.2039637267589569, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.13597580790519714, "reward_total_composite_mean": 0.6159030199050903, "reward_total_composite_std": 0.13597580790519714, "reward_total_mean": 0.6159030199050903, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9238545894622803, "rewards/meter/std": 0.2039637267589569, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6159030199050903, "rewards/total_composite/std": 0.13597580790519714, "sampling/importance_sampling_ratio/max": 1.6532901525497437, "sampling/importance_sampling_ratio/mean": 0.9989848136901855, "sampling/importance_sampling_ratio/min": 0.3318580687046051, "sampling/sampling_logp_difference/max": 1.1030478477478027, "sampling/sampling_logp_difference/mean": 0.00628118310123682, "step": 835 }, { "clip_ratio/high_max": 0.007069471699651331, "clip_ratio/high_mean": 0.007069471699651331, "clip_ratio/low_mean": 0.001773406460415572, "clip_ratio/low_min": 0.001773406460415572, "clip_ratio/region_mean": 0.008842878160066903, "completions/clipped_ratio": 0.0, "completions/max_length": 150.0, "completions/max_terminated_length": 150.0, "completions/mean_length": 142.25, "completions/mean_terminated_length": 142.25, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.044805840123444796, "epoch": 0.033578342772221555, "frac_reward_zero_std": 0.0, "grad_norm": 2.8023297786712646, "learning_rate": 7.46969696969697e-06, "loss": -0.0007, "num_tokens": 1856194.0, "reward": 0.7131145596504211, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8496776819229126, "reward_meter_std": 0.3326415419578552, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.29205048084259033, "reward_total_composite_mean": 0.7131145596504211, "reward_total_composite_std": 0.2920505106449127, "reward_total_mean": 0.7131145596504211, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8496776819229126, "rewards/meter/std": 0.3326415419578552, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7131145596504211, "rewards/total_composite/std": 0.2920505106449127, "sampling/importance_sampling_ratio/max": 1.5496021509170532, "sampling/importance_sampling_ratio/mean": 0.9984580278396606, "sampling/importance_sampling_ratio/min": 0.2702457010746002, "sampling/sampling_logp_difference/max": 1.3084237575531006, "sampling/sampling_logp_difference/mean": 0.012105772271752357, "step": 836 }, { "clip_ratio/high_max": 0.011363636702299118, "clip_ratio/high_mean": 0.011363636702299118, "clip_ratio/low_mean": 0.01515151560306549, "clip_ratio/low_min": 0.01515151560306549, "clip_ratio/region_mean": 0.02651515230536461, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 33.0, "completions/mean_terminated_length": 33.0, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.0944258663803339, "epoch": 0.03361850825400651, "frac_reward_zero_std": 0.0, "grad_norm": 11.809463500976562, "learning_rate": 7.4666666666666675e-06, "loss": 0.0053, "num_tokens": 1857618.0, "reward": 0.9952504634857178, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952504634857178, "reward_meter_std": 0.0004852505517192185, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004852571291849017, "reward_total_composite_mean": 0.9952504634857178, "reward_total_composite_std": 0.0004852505517192185, "reward_total_mean": 0.9952504634857178, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952504634857178, "rewards/meter/std": 0.0004852505517192185, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952504634857178, "rewards/total_composite/std": 0.0004852505517192185, "sampling/importance_sampling_ratio/max": 1.5762113332748413, "sampling/importance_sampling_ratio/mean": 1.0051133632659912, "sampling/importance_sampling_ratio/min": 0.4723230004310608, "sampling/sampling_logp_difference/max": 0.7500922679901123, "sampling/sampling_logp_difference/mean": 0.017783869057893753, "step": 837 }, { "clip_ratio/high_max": 0.004315092111937702, "clip_ratio/high_mean": 0.004315092111937702, "clip_ratio/low_mean": 0.0017137863032985479, "clip_ratio/low_min": 0.0017137863032985479, "clip_ratio/region_mean": 0.0060288784152362496, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 295.375, "completions/mean_terminated_length": 295.375, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "entropy": 0.02827131818048656, "epoch": 0.03365867373579146, "frac_reward_zero_std": 0.0, "grad_norm": 2.168991804122925, "learning_rate": 7.463636363636364e-06, "loss": 0.0061, "num_tokens": 1861717.0, "reward": 0.4079222083091736, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.06469365209341049, "reward_meter_mean": 0.869155764579773, "reward_meter_std": 0.35062167048454285, "reward_repeat_penalty_mean": 0.5626935362815857, "reward_repeat_penalty_std": 0.04896574467420578, "reward_std": 0.170829638838768, "reward_total_composite_mean": 0.4079222083091736, "reward_total_composite_std": 0.1708296537399292, "reward_total_mean": 0.4079222083091736, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/meter/mean": 0.869155764579773, "rewards/meter/std": 0.35062167048454285, "rewards/repeat_penalty/mean": 0.5626935362815857, "rewards/repeat_penalty/std": 0.04896574467420578, "rewards/total_composite/mean": 0.4079222083091736, "rewards/total_composite/std": 0.1708296537399292, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9985241889953613, "sampling/importance_sampling_ratio/min": 0.15722240507602692, "sampling/sampling_logp_difference/max": 1.8500938415527344, "sampling/sampling_logp_difference/mean": 0.009027719497680664, "step": 838 }, { "clip_ratio/high_max": 0.0022935778833925724, "clip_ratio/high_mean": 0.0022935778833925724, "clip_ratio/low_mean": 0.006605345057323575, "clip_ratio/low_min": 0.006605345057323575, "clip_ratio/region_mean": 0.008898922940716147, "completions/clipped_ratio": 0.0, "completions/max_length": 119.0, "completions/max_terminated_length": 119.0, "completions/mean_length": 112.125, "completions/mean_terminated_length": 112.125, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.047560357954353094, "epoch": 0.033698839217576416, "frac_reward_zero_std": 0.0, "grad_norm": 2.164283275604248, "learning_rate": 7.460606060606061e-06, "loss": 0.0131, "num_tokens": 1863934.0, "reward": 0.795501172542572, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943764209747314, "reward_meter_std": 0.0010588886216282845, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008471118635497987, "reward_total_composite_mean": 0.795501172542572, "reward_total_composite_std": 0.0008471080800518394, "reward_total_mean": 0.795501172542572, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943764209747314, "rewards/meter/std": 0.0010588886216282845, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.795501172542572, "rewards/total_composite/std": 0.0008471080800518394, "sampling/importance_sampling_ratio/max": 1.7390761375427246, "sampling/importance_sampling_ratio/mean": 1.0025136470794678, "sampling/importance_sampling_ratio/min": 0.4561457335948944, "sampling/sampling_logp_difference/max": 0.7849429249763489, "sampling/sampling_logp_difference/mean": 0.008892908692359924, "step": 839 }, { "clip_ratio/high_max": 0.010228979168459773, "clip_ratio/high_mean": 0.010228979168459773, "clip_ratio/low_mean": 0.006410256493836641, "clip_ratio/low_min": 0.006410256493836641, "clip_ratio/region_mean": 0.016639235662296414, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 37.125, "completions/mean_terminated_length": 37.125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.11148730758577585, "epoch": 0.03373900469936137, "frac_reward_zero_std": 0.0, "grad_norm": 5.7492289543151855, "learning_rate": 7.4575757575757575e-06, "loss": 0.036, "num_tokens": 1865375.0, "reward": 0.7395036816596985, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7395036816596985, "reward_meter_std": 0.2953052818775177, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2953052818775177, "reward_total_composite_mean": 0.7395036816596985, "reward_total_composite_std": 0.2953052818775177, "reward_total_mean": 0.7395036816596985, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7395036816596985, "rewards/meter/std": 0.2953052818775177, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7395036816596985, "rewards/total_composite/std": 0.2953052818775177, "sampling/importance_sampling_ratio/max": 1.4985777139663696, "sampling/importance_sampling_ratio/mean": 1.0085185766220093, "sampling/importance_sampling_ratio/min": 0.6779606938362122, "sampling/sampling_logp_difference/max": 0.40451645851135254, "sampling/sampling_logp_difference/mean": 0.01924203522503376, "step": 840 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.005771921598352492, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 64.75, "completions/mean_terminated_length": 64.75, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.045040544122457504, "epoch": 0.033779170181146324, "frac_reward_zero_std": 0.0, "grad_norm": 3.028710126876831, "learning_rate": 7.454545454545456e-06, "loss": -0.0141, "num_tokens": 1867245.0, "reward": 0.9528952240943909, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9528952240943909, "reward_meter_std": 0.01815449446439743, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.018154479563236237, "reward_total_composite_mean": 0.9528952240943909, "reward_total_composite_std": 0.01815449446439743, "reward_total_mean": 0.9528952240943909, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9528952240943909, "rewards/meter/std": 0.01815449446439743, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9528952240943909, "rewards/total_composite/std": 0.01815449446439743, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0040351152420044, "sampling/importance_sampling_ratio/min": 0.5180370807647705, "sampling/sampling_logp_difference/max": 0.8644721508026123, "sampling/sampling_logp_difference/mean": 0.01176523882895708, "step": 841 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.009699730202555656, "clip_ratio/low_min": 0.009699730202555656, "clip_ratio/region_mean": 0.009699730202555656, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 37.0, "completions/mean_terminated_length": 37.0, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.04681507125496864, "epoch": 0.03381933566293128, "frac_reward_zero_std": 0.0, "grad_norm": 9.17272663116455, "learning_rate": 7.451515151515152e-06, "loss": 0.0297, "num_tokens": 1868749.0, "reward": 0.700322687625885, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.700322687625885, "reward_meter_std": 0.3828321695327759, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3828321397304535, "reward_total_composite_mean": 0.700322687625885, "reward_total_composite_std": 0.3828321695327759, "reward_total_mean": 0.700322687625885, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.700322687625885, "rewards/meter/std": 0.3828321695327759, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.700322687625885, "rewards/total_composite/std": 0.3828321695327759, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0074673891067505, "sampling/importance_sampling_ratio/min": 0.6239942312240601, "sampling/sampling_logp_difference/max": 0.964850664138794, "sampling/sampling_logp_difference/mean": 0.011855069547891617, "step": 842 }, { "clip_ratio/high_max": 0.0002934272342827171, "clip_ratio/high_mean": 0.0002934272342827171, "clip_ratio/low_mean": 0.0011018531513400376, "clip_ratio/low_min": 0.0011018531513400376, "clip_ratio/region_mean": 0.0013952803856227547, "completions/clipped_ratio": 0.0, "completions/max_length": 461.0, "completions/max_terminated_length": 461.0, "completions/mean_length": 452.625, "completions/mean_terminated_length": 452.625, "completions/min_length": 426.0, "completions/min_terminated_length": 426.0, "entropy": 0.01467959355795756, "epoch": 0.03385950114471623, "frac_reward_zero_std": 0.0, "grad_norm": 2.23907470703125, "learning_rate": 7.448484848484849e-06, "loss": 0.0211, "num_tokens": 1874562.0, "reward": 0.5178769826889038, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9270833730697632, "reward_count_adherence_std": 0.029462777078151703, "reward_meter_mean": 0.9962563514709473, "reward_meter_std": 0.001195572316646576, "reward_repeat_penalty_mean": 0.5606521368026733, "reward_repeat_penalty_std": 0.0018446200992912054, "reward_std": 0.018427973613142967, "reward_total_composite_mean": 0.5178769826889038, "reward_total_composite_std": 0.01842796988785267, "reward_total_mean": 0.5178769826889038, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9270833730697632, "rewards/count_adherence/std": 0.029462777078151703, "rewards/meter/mean": 0.9962563514709473, "rewards/meter/std": 0.001195572316646576, "rewards/repeat_penalty/mean": 0.5606521368026733, "rewards/repeat_penalty/std": 0.0018446200992912054, "rewards/total_composite/mean": 0.5178769826889038, "rewards/total_composite/std": 0.01842796988785267, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.999667763710022, "sampling/importance_sampling_ratio/min": 0.01644536294043064, "sampling/sampling_logp_difference/max": 4.1077117919921875, "sampling/sampling_logp_difference/mean": 0.004932690877467394, "step": 843 }, { "clip_ratio/high_max": 0.002457805967424065, "clip_ratio/high_mean": 0.002457805967424065, "clip_ratio/low_mean": 0.011086393264122307, "clip_ratio/low_min": 0.011086393264122307, "clip_ratio/region_mean": 0.013544199231546372, "completions/clipped_ratio": 0.0, "completions/max_length": 158.0, "completions/max_terminated_length": 158.0, "completions/mean_length": 149.5, "completions/mean_terminated_length": 149.5, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.09837253391742706, "epoch": 0.033899666626501186, "frac_reward_zero_std": 0.0, "grad_norm": 2.640986919403076, "learning_rate": 7.445454545454546e-06, "loss": -0.0116, "num_tokens": 1877374.0, "reward": 0.3916763961315155, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5467606782913208, "reward_meter_std": 0.4782540798187256, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.34031397104263306, "reward_total_composite_mean": 0.3916763961315155, "reward_total_composite_std": 0.34031400084495544, "reward_total_mean": 0.3916763961315155, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5467606782913208, "rewards/meter/std": 0.4782540798187256, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.3916763961315155, "rewards/total_composite/std": 0.34031400084495544, "sampling/importance_sampling_ratio/max": 1.7641745805740356, "sampling/importance_sampling_ratio/mean": 1.0008183717727661, "sampling/importance_sampling_ratio/min": 0.03349503502249718, "sampling/sampling_logp_difference/max": 3.396358013153076, "sampling/sampling_logp_difference/mean": 0.019425617530941963, "step": 844 }, { "clip_ratio/high_max": 0.017289764247834682, "clip_ratio/high_mean": 0.017289764247834682, "clip_ratio/low_mean": 0.01767799479421228, "clip_ratio/low_min": 0.01767799479421228, "clip_ratio/region_mean": 0.034967759042046964, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 71.875, "completions/mean_terminated_length": 71.875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.16639000456780195, "epoch": 0.03393983210828614, "frac_reward_zero_std": 0.0, "grad_norm": 6.265551567077637, "learning_rate": 7.442424242424243e-06, "loss": 0.0112, "num_tokens": 1879205.0, "reward": 0.5038976669311523, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.522850513458252, "reward_meter_std": 0.4167693555355072, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.42369258403778076, "reward_total_composite_mean": 0.5038976669311523, "reward_total_composite_std": 0.42369261384010315, "reward_total_mean": 0.5038976669311523, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.522850513458252, "rewards/meter/std": 0.4167693555355072, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.5038976669311523, "rewards/total_composite/std": 0.42369261384010315, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023622512817383, "sampling/importance_sampling_ratio/min": 0.2801474928855896, "sampling/sampling_logp_difference/max": 1.2724390029907227, "sampling/sampling_logp_difference/mean": 0.03976872190833092, "step": 845 }, { "clip_ratio/high_max": 0.0058011687360703945, "clip_ratio/high_mean": 0.0058011687360703945, "clip_ratio/low_mean": 0.0011160714784637094, "clip_ratio/low_min": 0.0011160714784637094, "clip_ratio/region_mean": 0.006917240214534104, "completions/clipped_ratio": 0.0, "completions/max_length": 336.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 321.875, "completions/mean_terminated_length": 321.875, "completions/min_length": 309.0, "completions/min_terminated_length": 309.0, "entropy": 0.03188518015667796, "epoch": 0.033979997590071094, "frac_reward_zero_std": 0.0, "grad_norm": 4.094456672668457, "learning_rate": 7.439393939393939e-06, "loss": -0.0053, "num_tokens": 1883732.0, "reward": 0.3497573733329773, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9861111044883728, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.6540272831916809, "reward_meter_std": 0.39356982707977295, "reward_repeat_penalty_mean": 0.4677932560443878, "reward_repeat_penalty_std": 0.22017233073711395, "reward_std": 0.2734247148036957, "reward_total_composite_mean": 0.3497573733329773, "reward_total_composite_std": 0.2734247148036957, "reward_total_mean": 0.3497573733329773, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9861111044883728, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.6540272831916809, "rewards/meter/std": 0.39356982707977295, "rewards/repeat_penalty/mean": 0.4677932560443878, "rewards/repeat_penalty/std": 0.22017233073711395, "rewards/total_composite/mean": 0.3497573733329773, "rewards/total_composite/std": 0.2734247148036957, "sampling/importance_sampling_ratio/max": 1.768311619758606, "sampling/importance_sampling_ratio/mean": 0.9985901713371277, "sampling/importance_sampling_ratio/min": 0.15859971940517426, "sampling/sampling_logp_difference/max": 1.841371774673462, "sampling/sampling_logp_difference/mean": 0.007702399045228958, "step": 846 }, { "clip_ratio/high_max": 0.019930581096559763, "clip_ratio/high_mean": 0.019930581096559763, "clip_ratio/low_mean": 0.02969917980954051, "clip_ratio/low_min": 0.02969917980954051, "clip_ratio/region_mean": 0.04962976090610027, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 72.625, "completions/mean_terminated_length": 72.625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2849195022135973, "epoch": 0.03402016307185605, "frac_reward_zero_std": 0.0, "grad_norm": 6.007223606109619, "learning_rate": 7.4363636363636375e-06, "loss": 0.045, "num_tokens": 1885713.0, "reward": 0.2634860873222351, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.26388847827911377, "reward_meter_std": 0.3744986653327942, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.3748124837875366, "reward_total_composite_mean": 0.2634860873222351, "reward_total_composite_std": 0.3748124837875366, "reward_total_mean": 0.2634860873222351, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.26388847827911377, "rewards/meter/std": 0.3744986653327942, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.2634860873222351, "rewards/total_composite/std": 0.3748124837875366, "sampling/importance_sampling_ratio/max": 1.518433690071106, "sampling/importance_sampling_ratio/mean": 1.007526159286499, "sampling/importance_sampling_ratio/min": 0.2518492639064789, "sampling/sampling_logp_difference/max": 1.3789244890213013, "sampling/sampling_logp_difference/mean": 0.04990806803107262, "step": 847 }, { "clip_ratio/high_max": 0.013045326224528253, "clip_ratio/high_mean": 0.013045326224528253, "clip_ratio/low_mean": 0.005942982388660312, "clip_ratio/low_min": 0.005942982388660312, "clip_ratio/region_mean": 0.018988308613188565, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 106.125, "completions/mean_terminated_length": 106.125, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.08717937674373388, "epoch": 0.034060328553641, "frac_reward_zero_std": 0.0, "grad_norm": 2.973163366317749, "learning_rate": 7.433333333333334e-06, "loss": 0.0001, "num_tokens": 1888010.0, "reward": 0.5016140341758728, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6358721256256104, "reward_meter_std": 0.32246294617652893, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.2674255073070526, "reward_total_composite_mean": 0.5016140341758728, "reward_total_composite_std": 0.2674255073070526, "reward_total_mean": 0.5016140341758728, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6358721256256104, "rewards/meter/std": 0.32246294617652893, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.5016140341758728, "rewards/total_composite/std": 0.2674255073070526, "sampling/importance_sampling_ratio/max": 1.6591869592666626, "sampling/importance_sampling_ratio/mean": 0.9971781969070435, "sampling/importance_sampling_ratio/min": 0.1556275188922882, "sampling/sampling_logp_difference/max": 1.8602898120880127, "sampling/sampling_logp_difference/mean": 0.022296659648418427, "step": 848 }, { "clip_ratio/high_max": 0.00044169611646793783, "clip_ratio/high_mean": 0.00044169611646793783, "clip_ratio/low_mean": 0.00044014083687216043, "clip_ratio/low_min": 0.00044014083687216043, "clip_ratio/region_mean": 0.0008818369533400983, "completions/clipped_ratio": 0.0, "completions/max_length": 284.0, "completions/max_terminated_length": 284.0, "completions/mean_length": 283.75, "completions/mean_terminated_length": 283.75, "completions/min_length": 283.0, "completions/min_terminated_length": 283.0, "entropy": 0.010985135799273849, "epoch": 0.034100494035425956, "frac_reward_zero_std": 0.0, "grad_norm": 0.23050236701965332, "learning_rate": 7.430303030303031e-06, "loss": 0.001, "num_tokens": 1892024.0, "reward": 0.5979195833206177, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9965326189994812, "reward_meter_std": 0.00021112659305799752, "reward_repeat_penalty_mean": 0.6000000238418579, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000126663115224801, "reward_total_composite_mean": 0.5979195833206177, "reward_total_composite_std": 0.00012666928523685783, "reward_total_mean": 0.5979195833206177, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9965326189994812, "rewards/meter/std": 0.00021112659305799752, "rewards/repeat_penalty/mean": 0.6000000238418579, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5979195833206177, "rewards/total_composite/std": 0.00012666928523685783, "sampling/importance_sampling_ratio/max": 1.2486354112625122, "sampling/importance_sampling_ratio/mean": 1.0001165866851807, "sampling/importance_sampling_ratio/min": 0.4672318696975708, "sampling/sampling_logp_difference/max": 0.7609295845031738, "sampling/sampling_logp_difference/mean": 0.0016167466528713703, "step": 849 }, { "clip_ratio/high_max": 0.031166881788522005, "clip_ratio/high_mean": 0.031166881788522005, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/region_mean": 0.03830973897129297, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 35.5, "completions/mean_terminated_length": 35.5, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.15574337635189295, "epoch": 0.03414065951721091, "frac_reward_zero_std": 0.0, "grad_norm": 7.60121488571167, "learning_rate": 7.4272727272727275e-06, "loss": -0.0004, "num_tokens": 1893564.0, "reward": 0.8676612377166748, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9917958974838257, "reward_meter_std": 0.008829712867736816, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.350698858499527, "reward_total_composite_mean": 0.8676612377166748, "reward_total_composite_std": 0.35069888830184937, "reward_total_mean": 0.8676612377166748, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9917958974838257, "rewards/meter/std": 0.008829712867736816, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8676612377166748, "rewards/total_composite/std": 0.35069888830184937, "sampling/importance_sampling_ratio/max": 1.9962736368179321, "sampling/importance_sampling_ratio/mean": 1.0052940845489502, "sampling/importance_sampling_ratio/min": 0.03371865674853325, "sampling/sampling_logp_difference/max": 3.3897039890289307, "sampling/sampling_logp_difference/mean": 0.04963725060224533, "step": 850 }, { "epoch": 0.03414065951721091, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/max_length": 402.6923076923077, "eval_completions/max_terminated_length": 389.9230769230769, "eval_completions/mean_length": 209.85576923076923, "eval_completions/mean_terminated_length": 206.50412104679987, "eval_completions/min_length": 62.0, "eval_completions/min_terminated_length": 62.0, "eval_entropy": 0.046050483074325785, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 1893564.0, "eval_reward": 0.4121822485556969, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9535174920008733, "eval_reward_count_adherence_std": 0.07147554422800358, "eval_reward_meter_mean": 0.6430152883896461, "eval_reward_meter_std": 0.3854568119232471, "eval_reward_repeat_penalty_mean": 0.6578034116671636, "eval_reward_repeat_penalty_std": 0.21721924497531012, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.4121822485556969, "eval_reward_total_composite_std": 0.3158151931487597, "eval_reward_total_mean": 0.4121822485556969, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9535174920008733, "eval_rewards/count_adherence/std": 0.07147554422800358, "eval_rewards/meter/mean": 0.6430152883896461, "eval_rewards/meter/std": 0.3854568119232471, "eval_rewards/repeat_penalty/mean": 0.6578034116671636, "eval_rewards/repeat_penalty/std": 0.21721924497531012, "eval_rewards/total_composite/mean": 0.4121822485556969, "eval_rewards/total_composite/std": 0.3158151931487597, "eval_runtime": 75.7065, "eval_samples_per_second": 1.374, "eval_sampling/importance_sampling_ratio/max": 1.308758864035973, "eval_sampling/importance_sampling_ratio/mean": 1.001105854144463, "eval_sampling/importance_sampling_ratio/min": 0.4886489510536194, "eval_sampling/sampling_logp_difference/max": 0.7618373265633216, "eval_sampling/sampling_logp_difference/mean": 0.00518767936871602, "eval_steps_per_second": 0.172, "step": 850 }, { "clip_ratio/high_max": 0.00752471067244187, "clip_ratio/high_mean": 0.00752471067244187, "clip_ratio/low_mean": 0.0013513513840734959, "clip_ratio/low_min": 0.0013513513840734959, "clip_ratio/region_mean": 0.008876062056515366, "completions/clipped_ratio": 0.0, "completions/max_length": 199.0, "completions/max_terminated_length": 199.0, "completions/mean_length": 186.0, "completions/mean_terminated_length": 186.0, "completions/min_length": 178.0, "completions/min_terminated_length": 178.0, "entropy": 0.033908116864040494, "epoch": 0.034180824998995864, "frac_reward_zero_std": 0.0, "grad_norm": 2.013568162918091, "learning_rate": 7.424242424242425e-06, "loss": 0.0004, "num_tokens": 1896556.0, "reward": 0.5742474794387817, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.989942193031311, "reward_meter_std": 0.007896743714809418, "reward_repeat_penalty_mean": 0.5795454382896423, "reward_repeat_penalty_std": 0.16070608794689178, "reward_std": 0.16005171835422516, "reward_total_composite_mean": 0.5742474794387817, "reward_total_composite_std": 0.16005170345306396, "reward_total_mean": 0.5742474794387817, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.989942193031311, "rewards/meter/std": 0.007896743714809418, "rewards/repeat_penalty/mean": 0.5795454382896423, "rewards/repeat_penalty/std": 0.16070608794689178, "rewards/total_composite/mean": 0.5742474794387817, "rewards/total_composite/std": 0.16005170345306396, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0001335144042969, "sampling/importance_sampling_ratio/min": 0.3101670742034912, "sampling/sampling_logp_difference/max": 1.5690075159072876, "sampling/sampling_logp_difference/mean": 0.011838030070066452, "step": 851 }, { "clip_ratio/high_max": 0.006779896444641054, "clip_ratio/high_mean": 0.006779896444641054, "clip_ratio/low_mean": 0.006956097553484142, "clip_ratio/low_min": 0.006956097553484142, "clip_ratio/region_mean": 0.013735993998125196, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 72.5, "completions/mean_terminated_length": 72.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.10962006729096174, "epoch": 0.03422099048078082, "frac_reward_zero_std": 0.0, "grad_norm": 6.178599834442139, "learning_rate": 7.421212121212121e-06, "loss": -0.0116, "num_tokens": 1898384.0, "reward": 0.6031950116157532, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6383301019668579, "reward_meter_std": 0.34624141454696655, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.3757425546646118, "reward_total_composite_mean": 0.6031950116157532, "reward_total_composite_std": 0.3757425546646118, "reward_total_mean": 0.6031950116157532, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6383301019668579, "rewards/meter/std": 0.34624141454696655, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.6031950116157532, "rewards/total_composite/std": 0.3757425546646118, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9961123466491699, "sampling/importance_sampling_ratio/min": 0.11037729680538177, "sampling/sampling_logp_difference/max": 2.203850746154785, "sampling/sampling_logp_difference/mean": 0.03135836869478226, "step": 852 }, { "clip_ratio/high_max": 0.0017242564936168492, "clip_ratio/high_mean": 0.0017242564936168492, "clip_ratio/low_mean": 0.002888624498154968, "clip_ratio/low_min": 0.002888624498154968, "clip_ratio/region_mean": 0.004612880991771817, "completions/clipped_ratio": 0.0, "completions/max_length": 405.0, "completions/max_terminated_length": 405.0, "completions/mean_length": 373.75, "completions/mean_terminated_length": 373.75, "completions/min_length": 351.0, "completions/min_terminated_length": 351.0, "entropy": 0.03790131723508239, "epoch": 0.03426115596256577, "frac_reward_zero_std": 0.0, "grad_norm": 1.2332594394683838, "learning_rate": 7.4181818181818185e-06, "loss": 0.0349, "num_tokens": 1903150.0, "reward": 0.2684391736984253, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7692307829856873, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6268174648284912, "reward_meter_std": 0.39189305901527405, "reward_repeat_penalty_mean": 0.5203947424888611, "reward_repeat_penalty_std": 0.10010629892349243, "reward_std": 0.18215534090995789, "reward_total_composite_mean": 0.2684391736984253, "reward_total_composite_std": 0.18215535581111908, "reward_total_mean": 0.2684391736984253, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7692307829856873, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6268174648284912, "rewards/meter/std": 0.39189305901527405, "rewards/repeat_penalty/mean": 0.5203947424888611, "rewards/repeat_penalty/std": 0.10010629892349243, "rewards/total_composite/mean": 0.2684391736984253, "rewards/total_composite/std": 0.18215535581111908, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0005637407302856, "sampling/importance_sampling_ratio/min": 0.1743081957101822, "sampling/sampling_logp_difference/max": 1.788743495941162, "sampling/sampling_logp_difference/mean": 0.00892745703458786, "step": 853 }, { "clip_ratio/high_max": 0.008279926958493888, "clip_ratio/high_mean": 0.008279926958493888, "clip_ratio/low_mean": 0.0039122195448726416, "clip_ratio/low_min": 0.0039122195448726416, "clip_ratio/region_mean": 0.01219214650336653, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 125.75, "completions/mean_terminated_length": 125.75, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "entropy": 0.06485871458426118, "epoch": 0.034301321444350726, "frac_reward_zero_std": 0.0, "grad_norm": 3.387570858001709, "learning_rate": 7.415151515151515e-06, "loss": 0.023, "num_tokens": 1905516.0, "reward": 0.5973337292671204, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9104515314102173, "reward_meter_std": 0.12032636255025864, "reward_repeat_penalty_mean": 0.6607142686843872, "reward_repeat_penalty_std": 0.15152288973331451, "reward_std": 0.1518002152442932, "reward_total_composite_mean": 0.5973337292671204, "reward_total_composite_std": 0.1518002152442932, "reward_total_mean": 0.5973337292671204, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9104515314102173, "rewards/meter/std": 0.12032636255025864, "rewards/repeat_penalty/mean": 0.6607142686843872, "rewards/repeat_penalty/std": 0.15152288973331451, "rewards/total_composite/mean": 0.5973337292671204, "rewards/total_composite/std": 0.1518002152442932, "sampling/importance_sampling_ratio/max": 1.784404993057251, "sampling/importance_sampling_ratio/mean": 0.9986640214920044, "sampling/importance_sampling_ratio/min": 0.09487178921699524, "sampling/sampling_logp_difference/max": 2.355228900909424, "sampling/sampling_logp_difference/mean": 0.014981008134782314, "step": 854 }, { "clip_ratio/high_max": 0.01156850962433964, "clip_ratio/high_mean": 0.01156850962433964, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/region_mean": 0.015536763821728528, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 64.875, "completions/mean_terminated_length": 64.875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.06852800818160176, "epoch": 0.03434148692613568, "frac_reward_zero_std": 0.0, "grad_norm": 5.122466564178467, "learning_rate": 7.412121212121213e-06, "loss": -0.006, "num_tokens": 1907251.0, "reward": 0.9497148990631104, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9911971092224121, "reward_meter_std": 0.0026730033569037914, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11557891964912415, "reward_total_composite_mean": 0.9497148990631104, "reward_total_composite_std": 0.11557892709970474, "reward_total_mean": 0.9497148990631104, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9911971092224121, "rewards/meter/std": 0.0026730033569037914, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9497148990631104, "rewards/total_composite/std": 0.11557892709970474, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9993307590484619, "sampling/importance_sampling_ratio/min": 0.32319414615631104, "sampling/sampling_logp_difference/max": 1.6212968826293945, "sampling/sampling_logp_difference/mean": 0.021537287160754204, "step": 855 }, { "clip_ratio/high_max": 0.011599511839449406, "clip_ratio/high_mean": 0.011599511839449406, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/region_mean": 0.013522588764317334, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.75, "completions/mean_terminated_length": 64.75, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.07705914694815874, "epoch": 0.03438165240792063, "frac_reward_zero_std": 0.0, "grad_norm": 6.793356418609619, "learning_rate": 7.40909090909091e-06, "loss": 0.0002, "num_tokens": 1909129.0, "reward": 0.9881373643875122, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9881373643875122, "reward_meter_std": 0.009240454062819481, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00924046989530325, "reward_total_composite_mean": 0.9881373643875122, "reward_total_composite_std": 0.009240454062819481, "reward_total_mean": 0.9881373643875122, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9881373643875122, "rewards/meter/std": 0.009240454062819481, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9881373643875122, "rewards/total_composite/std": 0.009240454062819481, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9964679479598999, "sampling/importance_sampling_ratio/min": 0.3201471269130707, "sampling/sampling_logp_difference/max": 1.138974666595459, "sampling/sampling_logp_difference/mean": 0.017327219247817993, "step": 856 }, { "clip_ratio/high_max": 0.018700583139434457, "clip_ratio/high_mean": 0.018700583139434457, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.018700583139434457, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 75.75, "completions/mean_terminated_length": 75.75, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.07773020630702376, "epoch": 0.03442181788970559, "frac_reward_zero_std": 0.0, "grad_norm": 3.6093900203704834, "learning_rate": 7.406060606060607e-06, "loss": 0.0072, "num_tokens": 1910935.0, "reward": 0.9496715664863586, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.991020679473877, "reward_meter_std": 0.00623312359675765, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11656944453716278, "reward_total_composite_mean": 0.9496715664863586, "reward_total_composite_std": 0.11656945198774338, "reward_total_mean": 0.9496715664863586, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.991020679473877, "rewards/meter/std": 0.00623312359675765, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9496715664863586, "rewards/total_composite/std": 0.11656945198774338, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9909502863883972, "sampling/importance_sampling_ratio/min": 0.03157224878668785, "sampling/sampling_logp_difference/max": 3.455476760864258, "sampling/sampling_logp_difference/mean": 0.041222382336854935, "step": 857 }, { "clip_ratio/high_max": 0.006283877417445183, "clip_ratio/high_mean": 0.006283877417445183, "clip_ratio/low_mean": 0.009092815220355988, "clip_ratio/low_min": 0.009092815220355988, "clip_ratio/region_mean": 0.01537669263780117, "completions/clipped_ratio": 0.0, "completions/max_length": 268.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 254.625, "completions/mean_terminated_length": 254.625, "completions/min_length": 233.0, "completions/min_terminated_length": 233.0, "entropy": 0.1408333508297801, "epoch": 0.03446198337149054, "frac_reward_zero_std": 0.0, "grad_norm": 1.946842908859253, "learning_rate": 7.403030303030304e-06, "loss": 0.0285, "num_tokens": 1914420.0, "reward": 0.3126515746116638, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6067160367965698, "reward_meter_std": 0.3461027145385742, "reward_repeat_penalty_mean": 0.5166666507720947, "reward_repeat_penalty_std": 0.17728106677532196, "reward_std": 0.2052178531885147, "reward_total_composite_mean": 0.3126515746116638, "reward_total_composite_std": 0.2052178531885147, "reward_total_mean": 0.3126515746116638, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6067160367965698, "rewards/meter/std": 0.3461027145385742, "rewards/repeat_penalty/mean": 0.5166666507720947, "rewards/repeat_penalty/std": 0.17728106677532196, "rewards/total_composite/mean": 0.3126515746116638, "rewards/total_composite/std": 0.2052178531885147, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0007023811340332, "sampling/importance_sampling_ratio/min": 0.26724570989608765, "sampling/sampling_logp_difference/max": 1.3195867538452148, "sampling/sampling_logp_difference/mean": 0.020115505903959274, "step": 858 }, { "clip_ratio/high_max": 0.008425170031841844, "clip_ratio/high_mean": 0.008425170031841844, "clip_ratio/low_mean": 0.003665413591079414, "clip_ratio/low_min": 0.003665413591079414, "clip_ratio/region_mean": 0.012090583622921258, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 136.5, "completions/mean_terminated_length": 136.5, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.048957847990095615, "epoch": 0.034502148853275495, "frac_reward_zero_std": 0.0, "grad_norm": 1.447547435760498, "learning_rate": 7.4e-06, "loss": 0.0073, "num_tokens": 1916992.0, "reward": 0.5436913371086121, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9829227924346924, "reward_meter_std": 0.003464686218649149, "reward_repeat_penalty_mean": 0.5535714626312256, "reward_repeat_penalty_std": 0.23458294570446014, "reward_std": 0.22969430685043335, "reward_total_composite_mean": 0.5436913371086121, "reward_total_composite_std": 0.22969432175159454, "reward_total_mean": 0.5436913371086121, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9829227924346924, "rewards/meter/std": 0.003464686218649149, "rewards/repeat_penalty/mean": 0.5535714626312256, "rewards/repeat_penalty/std": 0.23458294570446014, "rewards/total_composite/mean": 0.5436913371086121, "rewards/total_composite/std": 0.22969432175159454, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0010943412780762, "sampling/importance_sampling_ratio/min": 0.30878695845603943, "sampling/sampling_logp_difference/max": 1.1751036643981934, "sampling/sampling_logp_difference/mean": 0.012742106802761555, "step": 859 }, { "clip_ratio/high_max": 0.005814186413772404, "clip_ratio/high_mean": 0.005814186413772404, "clip_ratio/low_mean": 0.002871762844733894, "clip_ratio/low_min": 0.002871762844733894, "clip_ratio/region_mean": 0.008685949258506298, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 128.25, "completions/mean_terminated_length": 128.25, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.03678457555361092, "epoch": 0.03454231433506045, "frac_reward_zero_std": 0.0, "grad_norm": 1.1458154916763306, "learning_rate": 7.396969696969698e-06, "loss": 0.0014, "num_tokens": 1919482.0, "reward": 0.6316357851028442, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9833224415779114, "reward_meter_std": 0.010524172335863113, "reward_repeat_penalty_mean": 0.6428571939468384, "reward_repeat_penalty_std": 0.15272071957588196, "reward_std": 0.14875362813472748, "reward_total_composite_mean": 0.6316357851028442, "reward_total_composite_std": 0.14875362813472748, "reward_total_mean": 0.6316357851028442, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9833224415779114, "rewards/meter/std": 0.010524172335863113, "rewards/repeat_penalty/mean": 0.6428571939468384, "rewards/repeat_penalty/std": 0.15272071957588196, "rewards/total_composite/mean": 0.6316357851028442, "rewards/total_composite/std": 0.14875362813472748, "sampling/importance_sampling_ratio/max": 1.9968863725662231, "sampling/importance_sampling_ratio/mean": 1.0009667873382568, "sampling/importance_sampling_ratio/min": 0.29379504919052124, "sampling/sampling_logp_difference/max": 1.2248728275299072, "sampling/sampling_logp_difference/mean": 0.011860202066600323, "step": 860 }, { "clip_ratio/high_max": 0.007541382568888366, "clip_ratio/high_mean": 0.007541382568888366, "clip_ratio/low_mean": 0.002939337049610913, "clip_ratio/low_min": 0.002939337049610913, "clip_ratio/region_mean": 0.010480719618499279, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 130.75, "completions/mean_terminated_length": 130.75, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.034579032799229026, "epoch": 0.0345824798168454, "frac_reward_zero_std": 0.0, "grad_norm": 3.195672035217285, "learning_rate": 7.393939393939395e-06, "loss": -0.0084, "num_tokens": 1921968.0, "reward": 0.6525848507881165, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9875681400299072, "reward_meter_std": 0.003578658914193511, "reward_repeat_penalty_mean": 0.660714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07386582344770432, "reward_total_composite_mean": 0.6525848507881165, "reward_total_composite_std": 0.07386582344770432, "reward_total_mean": 0.6525848507881165, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9875681400299072, "rewards/meter/std": 0.003578658914193511, "rewards/repeat_penalty/mean": 0.660714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.6525848507881165, "rewards/total_composite/std": 0.07386582344770432, "sampling/importance_sampling_ratio/max": 1.8728368282318115, "sampling/importance_sampling_ratio/mean": 0.999982476234436, "sampling/importance_sampling_ratio/min": 0.26445281505584717, "sampling/sampling_logp_difference/max": 1.330092430114746, "sampling/sampling_logp_difference/mean": 0.009233459830284119, "step": 861 }, { "clip_ratio/high_max": 0.010707754408940673, "clip_ratio/high_mean": 0.010707754408940673, "clip_ratio/low_mean": 0.0039816161734052, "clip_ratio/low_min": 0.0039816161734052, "clip_ratio/region_mean": 0.014689370582345873, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 137.0, "completions/mean_terminated_length": 137.0, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.0997177753597498, "epoch": 0.03462264529863036, "frac_reward_zero_std": 0.0, "grad_norm": 4.800323963165283, "learning_rate": 7.390909090909092e-06, "loss": -0.026, "num_tokens": 1924096.0, "reward": 0.5928546190261841, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8122666478157043, "reward_meter_std": 0.3459046185016632, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.1322600096464157, "reward_std": 0.29506081342697144, "reward_total_composite_mean": 0.5928546190261841, "reward_total_composite_std": 0.2950608432292938, "reward_total_mean": 0.5928546190261841, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8122666478157043, "rewards/meter/std": 0.3459046185016632, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.1322600096464157, "rewards/total_composite/mean": 0.5928546190261841, "rewards/total_composite/std": 0.2950608432292938, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0024539232254028, "sampling/importance_sampling_ratio/min": 0.2477279007434845, "sampling/sampling_logp_difference/max": 1.3954243659973145, "sampling/sampling_logp_difference/mean": 0.024774322286248207, "step": 862 }, { "clip_ratio/high_max": 0.0028625954873859882, "clip_ratio/high_mean": 0.0028625954873859882, "clip_ratio/low_mean": 0.009212537202984095, "clip_ratio/low_min": 0.009212537202984095, "clip_ratio/region_mean": 0.012075132690370083, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 147.875, "completions/mean_terminated_length": 147.875, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.036845392314717174, "epoch": 0.03466281078041531, "frac_reward_zero_std": 0.0, "grad_norm": 3.5590970516204834, "learning_rate": 7.3878787878787885e-06, "loss": 0.0474, "num_tokens": 1926759.0, "reward": 0.6657319068908691, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.9950444102287292, "reward_meter_std": 0.0004635074583347887, "reward_repeat_penalty_mean": 0.6904761791229248, "reward_repeat_penalty_std": 0.06734350323677063, "reward_std": 0.00668557733297348, "reward_total_composite_mean": 0.6657319068908691, "reward_total_composite_std": 0.006685574073344469, "reward_total_mean": 0.6657319068908691, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.9950444102287292, "rewards/meter/std": 0.0004635074583347887, "rewards/repeat_penalty/mean": 0.6904761791229248, "rewards/repeat_penalty/std": 0.06734350323677063, "rewards/total_composite/mean": 0.6657319068908691, "rewards/total_composite/std": 0.006685574073344469, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9978508353233337, "sampling/importance_sampling_ratio/min": 0.06431855261325836, "sampling/sampling_logp_difference/max": 2.7439072132110596, "sampling/sampling_logp_difference/mean": 0.01787300407886505, "step": 863 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 76.0, "completions/mean_terminated_length": 76.0, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.010270603699609637, "epoch": 0.034702976262200265, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.384848484848486e-06, "loss": 0.0, "num_tokens": 1928559.0, "reward": 0.9935055375099182, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9935055375099182, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9935055375099182, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9935055375099182, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9935055375099182, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9935055375099182, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0238041877746582, "sampling/importance_sampling_ratio/mean": 1.0007295608520508, "sampling/importance_sampling_ratio/min": 0.9370561838150024, "sampling/sampling_logp_difference/max": 0.06501199305057526, "sampling/sampling_logp_difference/mean": 0.0010724789462983608, "step": 864 }, { "clip_ratio/high_max": 0.00829179841093719, "clip_ratio/high_mean": 0.00829179841093719, "clip_ratio/low_mean": 0.0016891892300918698, "clip_ratio/low_min": 0.0016891892300918698, "clip_ratio/region_mean": 0.00998098764102906, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 74.625, "completions/mean_terminated_length": 74.625, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.07608733791857958, "epoch": 0.03474314174398522, "frac_reward_zero_std": 0.0, "grad_norm": 3.38623309135437, "learning_rate": 7.381818181818182e-06, "loss": -0.002, "num_tokens": 1930412.0, "reward": 0.995256781578064, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.995256781578064, "reward_meter_std": 0.0008067663875408471, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008067605085670948, "reward_total_composite_mean": 0.995256781578064, "reward_total_composite_std": 0.0008067663875408471, "reward_total_mean": 0.995256781578064, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.995256781578064, "rewards/meter/std": 0.0008067663875408471, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.995256781578064, "rewards/total_composite/std": 0.0008067663875408471, "sampling/importance_sampling_ratio/max": 1.6714001893997192, "sampling/importance_sampling_ratio/mean": 0.9971731305122375, "sampling/importance_sampling_ratio/min": 0.14737869799137115, "sampling/sampling_logp_difference/max": 1.9147498607635498, "sampling/sampling_logp_difference/mean": 0.016164077445864677, "step": 865 }, { "clip_ratio/high_max": 0.004934210563078523, "clip_ratio/high_mean": 0.004934210563078523, "clip_ratio/low_mean": 0.0022522523067891598, "clip_ratio/low_min": 0.0022522523067891598, "clip_ratio/region_mean": 0.0071864628698676825, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 79.875, "completions/mean_terminated_length": 79.875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.029443061910569668, "epoch": 0.03478330722577017, "frac_reward_zero_std": 0.0, "grad_norm": 4.458425045013428, "learning_rate": 7.378787878787879e-06, "loss": 0.1348, "num_tokens": 1932315.0, "reward": 0.8895903825759888, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9195418953895569, "reward_meter_std": 0.21018952131271362, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.2949047088623047, "reward_total_composite_mean": 0.8895903825759888, "reward_total_composite_std": 0.2949047088623047, "reward_total_mean": 0.8895903825759888, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9195418953895569, "rewards/meter/std": 0.21018952131271362, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8895903825759888, "rewards/total_composite/std": 0.2949047088623047, "sampling/importance_sampling_ratio/max": 1.7624181509017944, "sampling/importance_sampling_ratio/mean": 1.0000438690185547, "sampling/importance_sampling_ratio/min": 0.17681357264518738, "sampling/sampling_logp_difference/max": 1.7326593399047852, "sampling/sampling_logp_difference/mean": 0.011505667120218277, "step": 866 }, { "clip_ratio/high_max": 0.010424387757666409, "clip_ratio/high_mean": 0.010424387757666409, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.012739202589727938, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 98.125, "completions/mean_terminated_length": 98.125, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.05557233979925513, "epoch": 0.03482347270755513, "frac_reward_zero_std": 0.0, "grad_norm": 2.170355796813965, "learning_rate": 7.375757575757576e-06, "loss": 0.0392, "num_tokens": 1934484.0, "reward": 0.7012972831726074, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8766216039657593, "reward_meter_std": 0.1744304597377777, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.13954436779022217, "reward_total_composite_mean": 0.7012972831726074, "reward_total_composite_std": 0.13954438269138336, "reward_total_mean": 0.7012972831726074, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8766216039657593, "rewards/meter/std": 0.1744304597377777, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7012972831726074, "rewards/total_composite/std": 0.13954438269138336, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000335454940796, "sampling/importance_sampling_ratio/min": 0.07385222613811493, "sampling/sampling_logp_difference/max": 2.60568904876709, "sampling/sampling_logp_difference/mean": 0.016189197078347206, "step": 867 }, { "clip_ratio/high_max": 0.002336448524147272, "clip_ratio/high_mean": 0.002336448524147272, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/region_mean": 0.00696607818827033, "completions/clipped_ratio": 0.0, "completions/max_length": 113.0, "completions/max_terminated_length": 113.0, "completions/mean_length": 110.875, "completions/mean_terminated_length": 110.875, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.028168250457383692, "epoch": 0.03486363818934008, "frac_reward_zero_std": 0.0, "grad_norm": 1.8654179573059082, "learning_rate": 7.372727272727274e-06, "loss": -0.0072, "num_tokens": 1936683.0, "reward": 0.712011456489563, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8900142908096313, "reward_meter_std": 0.2951919138431549, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.23615355789661407, "reward_total_composite_mean": 0.712011456489563, "reward_total_composite_std": 0.23615355789661407, "reward_total_mean": 0.712011456489563, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8900142908096313, "rewards/meter/std": 0.2951919138431549, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.712011456489563, "rewards/total_composite/std": 0.23615355789661407, "sampling/importance_sampling_ratio/max": 1.3221184015274048, "sampling/importance_sampling_ratio/mean": 1.0000348091125488, "sampling/importance_sampling_ratio/min": 0.36075422167778015, "sampling/sampling_logp_difference/max": 1.0195584297180176, "sampling/sampling_logp_difference/mean": 0.004938235506415367, "step": 868 }, { "clip_ratio/high_max": 0.010045865317806602, "clip_ratio/high_mean": 0.010045865317806602, "clip_ratio/low_mean": 0.008417388889938593, "clip_ratio/low_min": 0.008417388889938593, "clip_ratio/region_mean": 0.018463254207745194, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 107.25, "completions/mean_terminated_length": 107.25, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.12705577816814184, "epoch": 0.034903803671125035, "frac_reward_zero_std": 0.0, "grad_norm": 3.4994313716888428, "learning_rate": 7.36969696969697e-06, "loss": -0.019, "num_tokens": 1938765.0, "reward": 0.8894296884536743, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9882686138153076, "reward_meter_std": 0.002476355992257595, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.10556221753358841, "reward_total_composite_mean": 0.8894296884536743, "reward_total_composite_std": 0.10556221753358841, "reward_total_mean": 0.8894296884536743, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9882686138153076, "rewards/meter/std": 0.002476355992257595, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8894296884536743, "rewards/total_composite/std": 0.10556221753358841, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004834771156311, "sampling/importance_sampling_ratio/min": 0.21899887919425964, "sampling/sampling_logp_difference/max": 1.518688678741455, "sampling/sampling_logp_difference/mean": 0.02293246239423752, "step": 869 }, { "clip_ratio/high_max": 0.015459735121112317, "clip_ratio/high_mean": 0.015459735121112317, "clip_ratio/low_mean": 0.0017879948718473315, "clip_ratio/low_min": 0.0017879948718473315, "clip_ratio/region_mean": 0.01724772999295965, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 138.375, "completions/mean_terminated_length": 138.375, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.05070830974727869, "epoch": 0.03494396915290999, "frac_reward_zero_std": 0.0, "grad_norm": 3.510559320449829, "learning_rate": 7.3666666666666676e-06, "loss": 0.0088, "num_tokens": 1941320.0, "reward": 0.6560331583023071, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983220100402832, "reward_meter_std": 0.0005191374220885336, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.052725713700056076, "reward_total_composite_mean": 0.6560331583023071, "reward_total_composite_std": 0.05272570997476578, "reward_total_mean": 0.6560331583023071, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983220100402832, "rewards/meter/std": 0.0005191374220885336, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.6560331583023071, "rewards/total_composite/std": 0.05272570997476578, "sampling/importance_sampling_ratio/max": 1.6303558349609375, "sampling/importance_sampling_ratio/mean": 0.9968522191047668, "sampling/importance_sampling_ratio/min": 0.22789107263088226, "sampling/sampling_logp_difference/max": 1.4788875579833984, "sampling/sampling_logp_difference/mean": 0.014895547181367874, "step": 870 }, { "clip_ratio/high_max": 0.009079249284695834, "clip_ratio/high_mean": 0.009079249284695834, "clip_ratio/low_mean": 0.0053914261516183615, "clip_ratio/low_min": 0.0053914261516183615, "clip_ratio/region_mean": 0.014470675436314195, "completions/clipped_ratio": 0.0, "completions/max_length": 201.0, "completions/max_terminated_length": 201.0, "completions/mean_length": 190.25, "completions/mean_terminated_length": 190.25, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.11951877502724528, "epoch": 0.03498413463469494, "frac_reward_zero_std": 0.0, "grad_norm": 1.7928707599639893, "learning_rate": 7.363636363636364e-06, "loss": -0.0038, "num_tokens": 1944242.0, "reward": 0.5602418184280396, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8820419907569885, "reward_meter_std": 0.2552332878112793, "reward_repeat_penalty_mean": 0.7638888955116272, "reward_repeat_penalty_std": 0.0927247703075409, "reward_std": 0.17721408605575562, "reward_total_composite_mean": 0.5602418184280396, "reward_total_composite_std": 0.1772141009569168, "reward_total_mean": 0.5602418184280396, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8820419907569885, "rewards/meter/std": 0.2552332878112793, "rewards/repeat_penalty/mean": 0.7638888955116272, "rewards/repeat_penalty/std": 0.0927247703075409, "rewards/total_composite/mean": 0.5602418184280396, "rewards/total_composite/std": 0.1772141009569168, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0031383037567139, "sampling/importance_sampling_ratio/min": 0.324779748916626, "sampling/sampling_logp_difference/max": 1.124608039855957, "sampling/sampling_logp_difference/mean": 0.015695709735155106, "step": 871 }, { "clip_ratio/high_max": 0.005582162644714117, "clip_ratio/high_mean": 0.005582162644714117, "clip_ratio/low_mean": 0.0038759689778089523, "clip_ratio/low_min": 0.0038759689778089523, "clip_ratio/region_mean": 0.00945813162252307, "completions/clipped_ratio": 0.0, "completions/max_length": 162.0, "completions/max_terminated_length": 162.0, "completions/mean_length": 144.75, "completions/mean_terminated_length": 144.75, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.05419332440942526, "epoch": 0.035024300116479896, "frac_reward_zero_std": 0.0, "grad_norm": 3.245631694793701, "learning_rate": 7.360606060606061e-06, "loss": -0.0316, "num_tokens": 1946920.0, "reward": 0.6906325221061707, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9848458170890808, "reward_meter_std": 0.03508731722831726, "reward_repeat_penalty_mean": 0.7250000238418579, "reward_repeat_penalty_std": 0.030304575338959694, "reward_std": 0.061346374452114105, "reward_total_composite_mean": 0.6906325221061707, "reward_total_composite_std": 0.06134637072682381, "reward_total_mean": 0.6906325221061707, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9848458170890808, "rewards/meter/std": 0.03508731722831726, "rewards/repeat_penalty/mean": 0.7250000238418579, "rewards/repeat_penalty/std": 0.030304575338959694, "rewards/total_composite/mean": 0.6906325221061707, "rewards/total_composite/std": 0.06134637072682381, "sampling/importance_sampling_ratio/max": 1.8771061897277832, "sampling/importance_sampling_ratio/mean": 1.0042208433151245, "sampling/importance_sampling_ratio/min": 0.14886555075645447, "sampling/sampling_logp_difference/max": 1.9047117233276367, "sampling/sampling_logp_difference/mean": 0.010910775512456894, "step": 872 }, { "clip_ratio/high_max": 0.027383167180232704, "clip_ratio/high_mean": 0.027383167180232704, "clip_ratio/low_mean": 0.013034759555011988, "clip_ratio/low_min": 0.013034759555011988, "clip_ratio/region_mean": 0.04041792673524469, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 64.875, "completions/mean_terminated_length": 64.875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.17148890905082226, "epoch": 0.03506446559826485, "frac_reward_zero_std": 0.0, "grad_norm": 5.7755126953125, "learning_rate": 7.357575757575758e-06, "loss": 0.0209, "num_tokens": 1948679.0, "reward": 0.9520989656448364, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9520989656448364, "reward_meter_std": 0.055099111050367355, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05509910732507706, "reward_total_composite_mean": 0.9520989656448364, "reward_total_composite_std": 0.055099111050367355, "reward_total_mean": 0.9520989656448364, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9520989656448364, "rewards/meter/std": 0.055099111050367355, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9520989656448364, "rewards/total_composite/std": 0.055099111050367355, "sampling/importance_sampling_ratio/max": 1.7566462755203247, "sampling/importance_sampling_ratio/mean": 1.0100115537643433, "sampling/importance_sampling_ratio/min": 0.2037644237279892, "sampling/sampling_logp_difference/max": 1.5907907485961914, "sampling/sampling_logp_difference/mean": 0.032992783933877945, "step": 873 }, { "clip_ratio/high_max": 0.007142857299186289, "clip_ratio/high_mean": 0.007142857299186289, "clip_ratio/low_mean": 0.0022123893722891808, "clip_ratio/low_min": 0.0022123893722891808, "clip_ratio/region_mean": 0.00935524667147547, "completions/clipped_ratio": 0.0, "completions/max_length": 113.0, "completions/max_terminated_length": 113.0, "completions/mean_length": 107.25, "completions/mean_terminated_length": 107.25, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.04265120159834623, "epoch": 0.035104631080049804, "frac_reward_zero_std": 0.0, "grad_norm": 1.7160921096801758, "learning_rate": 7.354545454545456e-06, "loss": 0.0149, "num_tokens": 1950913.0, "reward": 0.7964070439338684, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955088496208191, "reward_meter_std": 0.0011983213480561972, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009586560190655291, "reward_total_composite_mean": 0.7964070439338684, "reward_total_composite_std": 0.0009586615487933159, "reward_total_mean": 0.7964070439338684, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955088496208191, "rewards/meter/std": 0.0011983213480561972, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7964070439338684, "rewards/total_composite/std": 0.0009586615487933159, "sampling/importance_sampling_ratio/max": 1.3036043643951416, "sampling/importance_sampling_ratio/mean": 0.9997270703315735, "sampling/importance_sampling_ratio/min": 0.46094223856925964, "sampling/sampling_logp_difference/max": 0.7744824886322021, "sampling/sampling_logp_difference/mean": 0.008256379514932632, "step": 874 }, { "clip_ratio/high_max": 0.023883325862698257, "clip_ratio/high_mean": 0.023883325862698257, "clip_ratio/low_mean": 0.0026881720405071974, "clip_ratio/low_min": 0.0026881720405071974, "clip_ratio/region_mean": 0.026571497903205454, "completions/clipped_ratio": 0.0, "completions/max_length": 100.0, "completions/max_terminated_length": 100.0, "completions/mean_length": 94.125, "completions/mean_terminated_length": 94.125, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.07790858251973987, "epoch": 0.03514479656183476, "frac_reward_zero_std": 0.0, "grad_norm": 2.395378351211548, "learning_rate": 7.351515151515151e-06, "loss": -0.0005, "num_tokens": 1953074.0, "reward": 0.7692322134971619, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.992639422416687, "reward_meter_std": 0.006307392846792936, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06972822546958923, "reward_total_composite_mean": 0.7692322134971619, "reward_total_composite_std": 0.06972821056842804, "reward_total_mean": 0.7692322134971619, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.992639422416687, "rewards/meter/std": 0.006307392846792936, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7692322134971619, "rewards/total_composite/std": 0.06972821056842804, "sampling/importance_sampling_ratio/max": 1.5037858486175537, "sampling/importance_sampling_ratio/mean": 0.9954232573509216, "sampling/importance_sampling_ratio/min": 0.13049879670143127, "sampling/sampling_logp_difference/max": 2.036391258239746, "sampling/sampling_logp_difference/mean": 0.02180575206875801, "step": 875 }, { "clip_ratio/high_max": 0.0031779659911990166, "clip_ratio/high_mean": 0.0031779659911990166, "clip_ratio/low_mean": 0.007715584768448025, "clip_ratio/low_min": 0.007715584768448025, "clip_ratio/region_mean": 0.010893550759647042, "completions/clipped_ratio": 0.0, "completions/max_length": 236.0, "completions/max_terminated_length": 236.0, "completions/mean_length": 217.25, "completions/mean_terminated_length": 217.25, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.052224091021344066, "epoch": 0.03518496204361971, "frac_reward_zero_std": 0.0, "grad_norm": 2.0017435550689697, "learning_rate": 7.348484848484849e-06, "loss": -0.0419, "num_tokens": 1956524.0, "reward": 0.6128624677658081, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.9868931770324707, "reward_meter_std": 0.006763860583305359, "reward_repeat_penalty_mean": 0.6858974695205688, "reward_repeat_penalty_std": 0.011869488283991814, "reward_std": 0.02833697572350502, "reward_total_composite_mean": 0.6128624677658081, "reward_total_composite_std": 0.02833697944879532, "reward_total_mean": 0.6128624677658081, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.9868931770324707, "rewards/meter/std": 0.006763860583305359, "rewards/repeat_penalty/mean": 0.6858974695205688, "rewards/repeat_penalty/std": 0.011869488283991814, "rewards/total_composite/mean": 0.6128624677658081, "rewards/total_composite/std": 0.02833697944879532, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016402006149292, "sampling/importance_sampling_ratio/min": 0.29015299677848816, "sampling/sampling_logp_difference/max": 1.237346887588501, "sampling/sampling_logp_difference/mean": 0.012169601395726204, "step": 876 }, { "clip_ratio/high_max": 0.015677204821258783, "clip_ratio/high_mean": 0.015677204821258783, "clip_ratio/low_mean": 0.003247300977818668, "clip_ratio/low_min": 0.003247300977818668, "clip_ratio/region_mean": 0.01892450579907745, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 79.5, "completions/mean_terminated_length": 79.5, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.1384354429319501, "epoch": 0.035225127525404666, "frac_reward_zero_std": 0.0, "grad_norm": 6.4066290855407715, "learning_rate": 7.345454545454546e-06, "loss": -0.0079, "num_tokens": 1958416.0, "reward": 0.9876222610473633, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9876222610473633, "reward_meter_std": 0.013733073137700558, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01373306754976511, "reward_total_composite_mean": 0.9876222610473633, "reward_total_composite_std": 0.013733073137700558, "reward_total_mean": 0.9876222610473633, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9876222610473633, "rewards/meter/std": 0.013733073137700558, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9876222610473633, "rewards/total_composite/std": 0.013733073137700558, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0038245916366577, "sampling/importance_sampling_ratio/min": 0.24818849563598633, "sampling/sampling_logp_difference/max": 1.3935667276382446, "sampling/sampling_logp_difference/mean": 0.02791396901011467, "step": 877 }, { "clip_ratio/high_max": 0.014086603187024593, "clip_ratio/high_mean": 0.014086603187024593, "clip_ratio/low_mean": 0.016726079280488193, "clip_ratio/low_min": 0.016726079280488193, "clip_ratio/region_mean": 0.030812682467512786, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 69.375, "completions/mean_terminated_length": 69.375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.09143292810767889, "epoch": 0.03526529300718962, "frac_reward_zero_std": 0.0, "grad_norm": 3.125847816467285, "learning_rate": 7.342424242424243e-06, "loss": -0.0205, "num_tokens": 1960243.0, "reward": 0.9980708360671997, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980708360671997, "reward_meter_std": 0.00046432489762082696, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00046432187082245946, "reward_total_composite_mean": 0.9980708360671997, "reward_total_composite_std": 0.00046432489762082696, "reward_total_mean": 0.9980708360671997, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980708360671997, "rewards/meter/std": 0.00046432489762082696, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980708360671997, "rewards/total_composite/std": 0.00046432489762082696, "sampling/importance_sampling_ratio/max": 1.6681042909622192, "sampling/importance_sampling_ratio/mean": 0.999176561832428, "sampling/importance_sampling_ratio/min": 0.39883068203926086, "sampling/sampling_logp_difference/max": 0.9192183017730713, "sampling/sampling_logp_difference/mean": 0.021711641922593117, "step": 878 }, { "clip_ratio/high_max": 0.01865671609994024, "clip_ratio/high_mean": 0.01865671609994024, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.01865671609994024, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.125, "completions/mean_terminated_length": 65.125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.072424097917974, "epoch": 0.035305458488974574, "frac_reward_zero_std": 0.0, "grad_norm": 4.135697841644287, "learning_rate": 7.3393939393939395e-06, "loss": -0.0466, "num_tokens": 1962012.0, "reward": 0.7184880971908569, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7186502814292908, "reward_meter_std": 0.37441036105155945, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.37476426362991333, "reward_total_composite_mean": 0.7184880971908569, "reward_total_composite_std": 0.3747643232345581, "reward_total_mean": 0.7184880971908569, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7186502814292908, "rewards/meter/std": 0.37441036105155945, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.7184880971908569, "rewards/total_composite/std": 0.3747643232345581, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9979657530784607, "sampling/importance_sampling_ratio/min": 0.22408060729503632, "sampling/sampling_logp_difference/max": 1.4957494735717773, "sampling/sampling_logp_difference/mean": 0.025105303153395653, "step": 879 }, { "clip_ratio/high_max": 0.008080808212980628, "clip_ratio/high_mean": 0.008080808212980628, "clip_ratio/low_mean": 0.008119725622236729, "clip_ratio/low_min": 0.008119725622236729, "clip_ratio/region_mean": 0.016200533835217357, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 94.125, "completions/mean_terminated_length": 94.125, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.04804056999273598, "epoch": 0.03534562397075953, "frac_reward_zero_std": 0.0, "grad_norm": 4.517680644989014, "learning_rate": 7.336363636363637e-06, "loss": -0.0202, "num_tokens": 1964149.0, "reward": 0.28257015347480774, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3598088026046753, "reward_meter_std": 0.3960496783256531, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.31944531202316284, "reward_total_composite_mean": 0.28257015347480774, "reward_total_composite_std": 0.31944534182548523, "reward_total_mean": 0.28257015347480774, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3598088026046753, "rewards/meter/std": 0.3960496783256531, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.28257015347480774, "rewards/total_composite/std": 0.31944534182548523, "sampling/importance_sampling_ratio/max": 1.8587523698806763, "sampling/importance_sampling_ratio/mean": 0.9979188442230225, "sampling/importance_sampling_ratio/min": 0.3478960692882538, "sampling/sampling_logp_difference/max": 1.0558514595031738, "sampling/sampling_logp_difference/mean": 0.015417142771184444, "step": 880 }, { "clip_ratio/high_max": 0.002022653818130493, "clip_ratio/high_mean": 0.002022653818130493, "clip_ratio/low_mean": 0.003381519520189613, "clip_ratio/low_min": 0.003381519520189613, "clip_ratio/region_mean": 0.005404173338320106, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 292.625, "completions/mean_terminated_length": 292.625, "completions/min_length": 277.0, "completions/min_terminated_length": 277.0, "entropy": 0.04085008194670081, "epoch": 0.03538578945254448, "frac_reward_zero_std": 0.0, "grad_norm": 1.2314221858978271, "learning_rate": 7.333333333333333e-06, "loss": -0.0185, "num_tokens": 1968226.0, "reward": 0.4401511549949646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7386363744735718, "reward_count_adherence_std": 0.03214120864868164, "reward_meter_mean": 0.9959769248962402, "reward_meter_std": 0.0035888429265469313, "reward_repeat_penalty_mean": 0.5985294580459595, "reward_repeat_penalty_std": 0.004159451462328434, "reward_std": 0.014130760915577412, "reward_total_composite_mean": 0.4401511549949646, "reward_total_composite_std": 0.014130757190287113, "reward_total_mean": 0.4401511549949646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7386363744735718, "rewards/count_adherence/std": 0.03214120864868164, "rewards/meter/mean": 0.9959769248962402, "rewards/meter/std": 0.0035888429265469313, "rewards/repeat_penalty/mean": 0.5985294580459595, "rewards/repeat_penalty/std": 0.004159451462328434, "rewards/total_composite/mean": 0.4401511549949646, "rewards/total_composite/std": 0.014130757190287113, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0001860857009888, "sampling/importance_sampling_ratio/min": 0.3691328465938568, "sampling/sampling_logp_difference/max": 0.9965987205505371, "sampling/sampling_logp_difference/mean": 0.007294100243598223, "step": 881 }, { "clip_ratio/high_max": 0.0039639262249693274, "clip_ratio/high_mean": 0.0039639262249693274, "clip_ratio/low_mean": 0.0054536922834813595, "clip_ratio/low_min": 0.0054536922834813595, "clip_ratio/region_mean": 0.009417618508450687, "completions/clipped_ratio": 0.0, "completions/max_length": 166.0, "completions/max_terminated_length": 166.0, "completions/mean_length": 144.25, "completions/mean_terminated_length": 144.25, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.05786414351314306, "epoch": 0.035425954934329436, "frac_reward_zero_std": 0.0, "grad_norm": 3.3222203254699707, "learning_rate": 7.330303030303031e-06, "loss": -0.002, "num_tokens": 1970988.0, "reward": 0.5672966837882996, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.7575975060462952, "reward_meter_std": 0.32098710536956787, "reward_repeat_penalty_mean": 0.783730149269104, "reward_repeat_penalty_std": 0.05935349687933922, "reward_std": 0.2538285553455353, "reward_total_composite_mean": 0.5672966837882996, "reward_total_composite_std": 0.25382858514785767, "reward_total_mean": 0.5672966837882996, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.7575975060462952, "rewards/meter/std": 0.32098710536956787, "rewards/repeat_penalty/mean": 0.783730149269104, "rewards/repeat_penalty/std": 0.05935349687933922, "rewards/total_composite/mean": 0.5672966837882996, "rewards/total_composite/std": 0.25382858514785767, "sampling/importance_sampling_ratio/max": 1.9269946813583374, "sampling/importance_sampling_ratio/mean": 0.9993391633033752, "sampling/importance_sampling_ratio/min": 0.37250787019729614, "sampling/sampling_logp_difference/max": 0.987497091293335, "sampling/sampling_logp_difference/mean": 0.013289229944348335, "step": 882 }, { "clip_ratio/high_max": 0.02063214615918696, "clip_ratio/high_mean": 0.02063214615918696, "clip_ratio/low_mean": 0.00797146384138614, "clip_ratio/low_min": 0.00797146384138614, "clip_ratio/region_mean": 0.0286036100005731, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 61.125, "completions/mean_terminated_length": 61.125, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.1264942679554224, "epoch": 0.03546612041611439, "frac_reward_zero_std": 0.0, "grad_norm": 23.914064407348633, "learning_rate": 7.3272727272727285e-06, "loss": 0.0339, "num_tokens": 1972789.0, "reward": 0.9858500361442566, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9858500361442566, "reward_meter_std": 0.013770055025815964, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.013770047575235367, "reward_total_composite_mean": 0.9858500361442566, "reward_total_composite_std": 0.013770055025815964, "reward_total_mean": 0.9858500361442566, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9858500361442566, "rewards/meter/std": 0.013770055025815964, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9858500361442566, "rewards/total_composite/std": 0.013770055025815964, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0076637268066406, "sampling/importance_sampling_ratio/min": 0.030958153307437897, "sampling/sampling_logp_difference/max": 3.47511887550354, "sampling/sampling_logp_difference/mean": 0.058846138417720795, "step": 883 }, { "clip_ratio/high_max": 0.008322211273480207, "clip_ratio/high_mean": 0.008322211273480207, "clip_ratio/low_mean": 0.003500614082440734, "clip_ratio/low_min": 0.003500614082440734, "clip_ratio/region_mean": 0.01182282535592094, "completions/clipped_ratio": 0.0, "completions/max_length": 187.0, "completions/max_terminated_length": 187.0, "completions/mean_length": 178.625, "completions/mean_terminated_length": 178.625, "completions/min_length": 172.0, "completions/min_terminated_length": 172.0, "entropy": 0.05528515903279185, "epoch": 0.035506285897899344, "frac_reward_zero_std": 0.0, "grad_norm": 2.2200710773468018, "learning_rate": 7.324242424242425e-06, "loss": 0.0017, "num_tokens": 1975890.0, "reward": 0.39888328313827515, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6974109411239624, "reward_meter_std": 0.3949660360813141, "reward_repeat_penalty_mean": 0.6805555820465088, "reward_repeat_penalty_std": 0.03928370773792267, "reward_std": 0.23110367357730865, "reward_total_composite_mean": 0.39888328313827515, "reward_total_composite_std": 0.23110367357730865, "reward_total_mean": 0.39888328313827515, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6974109411239624, "rewards/meter/std": 0.3949660360813141, "rewards/repeat_penalty/mean": 0.6805555820465088, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.39888328313827515, "rewards/total_composite/std": 0.23110367357730865, "sampling/importance_sampling_ratio/max": 1.8987165689468384, "sampling/importance_sampling_ratio/mean": 0.9986303448677063, "sampling/importance_sampling_ratio/min": 0.12287382781505585, "sampling/sampling_logp_difference/max": 2.096597194671631, "sampling/sampling_logp_difference/mean": 0.015560545958578587, "step": 884 }, { "clip_ratio/high_max": 0.025362646207213402, "clip_ratio/high_mean": 0.025362646207213402, "clip_ratio/low_mean": 0.007144315168261528, "clip_ratio/low_min": 0.007144315168261528, "clip_ratio/region_mean": 0.03250696137547493, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 70.375, "completions/mean_terminated_length": 70.375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.10360199213027954, "epoch": 0.0355464513796843, "frac_reward_zero_std": 0.0, "grad_norm": 8.853222846984863, "learning_rate": 7.321212121212122e-06, "loss": 0.0102, "num_tokens": 1977749.0, "reward": 0.778793215751648, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.778793215751648, "reward_meter_std": 0.4060738682746887, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4060738682746887, "reward_total_composite_mean": 0.778793215751648, "reward_total_composite_std": 0.4060738682746887, "reward_total_mean": 0.778793215751648, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.778793215751648, "rewards/meter/std": 0.4060738682746887, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.778793215751648, "rewards/total_composite/std": 0.4060738682746887, "sampling/importance_sampling_ratio/max": 1.356745958328247, "sampling/importance_sampling_ratio/mean": 0.9921159744262695, "sampling/importance_sampling_ratio/min": 0.12295946478843689, "sampling/sampling_logp_difference/max": 2.095900535583496, "sampling/sampling_logp_difference/mean": 0.028910191729664803, "step": 885 }, { "clip_ratio/high_max": 0.002134879759978503, "clip_ratio/high_mean": 0.002134879759978503, "clip_ratio/low_mean": 0.0012175324372947216, "clip_ratio/low_min": 0.0012175324372947216, "clip_ratio/region_mean": 0.0033524121972732246, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 294.25, "completions/mean_terminated_length": 294.25, "completions/min_length": 291.0, "completions/min_terminated_length": 291.0, "entropy": 0.01867874281015247, "epoch": 0.03558661686146925, "frac_reward_zero_std": 0.0, "grad_norm": 2.6481733322143555, "learning_rate": 7.3181818181818186e-06, "loss": 0.0196, "num_tokens": 1982015.0, "reward": 0.45384329557418823, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8633161783218384, "reward_meter_std": 0.3173806965351105, "reward_repeat_penalty_mean": 0.5424836874008179, "reward_repeat_penalty_std": 0.1294051706790924, "reward_std": 0.1770940124988556, "reward_total_composite_mean": 0.45384329557418823, "reward_total_composite_std": 0.1770940124988556, "reward_total_mean": 0.45384329557418823, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8633161783218384, "rewards/meter/std": 0.3173806965351105, "rewards/repeat_penalty/mean": 0.5424836874008179, "rewards/repeat_penalty/std": 0.1294051706790924, "rewards/total_composite/mean": 0.45384329557418823, "rewards/total_composite/std": 0.1770940124988556, "sampling/importance_sampling_ratio/max": 1.9255647659301758, "sampling/importance_sampling_ratio/mean": 1.0008361339569092, "sampling/importance_sampling_ratio/min": 0.537617027759552, "sampling/sampling_logp_difference/max": 0.655219316482544, "sampling/sampling_logp_difference/mean": 0.003619949799031019, "step": 886 }, { "clip_ratio/high_max": 0.0046083927154541016, "clip_ratio/high_mean": 0.0046083927154541016, "clip_ratio/low_mean": 0.005965773249045014, "clip_ratio/low_min": 0.005965773249045014, "clip_ratio/region_mean": 0.010574165964499116, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 105.75, "completions/mean_terminated_length": 105.75, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.06286561442539096, "epoch": 0.035626782343254206, "frac_reward_zero_std": 0.0, "grad_norm": 2.381399393081665, "learning_rate": 7.315151515151516e-06, "loss": -0.0139, "num_tokens": 1984181.0, "reward": 0.8444880843162537, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9935408234596252, "reward_meter_std": 0.003044598735868931, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09176833927631378, "reward_total_composite_mean": 0.8444880843162537, "reward_total_composite_std": 0.09176833927631378, "reward_total_mean": 0.8444880843162537, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9935408234596252, "rewards/meter/std": 0.003044598735868931, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8444880843162537, "rewards/total_composite/std": 0.09176833927631378, "sampling/importance_sampling_ratio/max": 1.6447700262069702, "sampling/importance_sampling_ratio/mean": 1.0018950700759888, "sampling/importance_sampling_ratio/min": 0.4930652678012848, "sampling/sampling_logp_difference/max": 0.7071137428283691, "sampling/sampling_logp_difference/mean": 0.010191258043050766, "step": 887 }, { "clip_ratio/high_max": 0.006115591386333108, "clip_ratio/high_mean": 0.006115591386333108, "clip_ratio/low_mean": 0.003969253972172737, "clip_ratio/low_min": 0.003969253972172737, "clip_ratio/region_mean": 0.010084845358505845, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.0, "completions/mean_terminated_length": 62.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.08477870561182499, "epoch": 0.03566694782503916, "frac_reward_zero_std": 0.0, "grad_norm": 5.54829740524292, "learning_rate": 7.312121212121212e-06, "loss": 0.0134, "num_tokens": 1986101.0, "reward": 0.9698032140731812, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9698032140731812, "reward_meter_std": 0.021262098103761673, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.021262086927890778, "reward_total_composite_mean": 0.9698032140731812, "reward_total_composite_std": 0.021262098103761673, "reward_total_mean": 0.9698032140731812, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9698032140731812, "rewards/meter/std": 0.021262098103761673, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9698032140731812, "rewards/total_composite/std": 0.021262098103761673, "sampling/importance_sampling_ratio/max": 1.7581480741500854, "sampling/importance_sampling_ratio/mean": 1.0010225772857666, "sampling/importance_sampling_ratio/min": 0.38010627031326294, "sampling/sampling_logp_difference/max": 0.9673044681549072, "sampling/sampling_logp_difference/mean": 0.019981540739536285, "step": 888 }, { "clip_ratio/high_max": 0.005093896761536598, "clip_ratio/high_mean": 0.005093896761536598, "clip_ratio/low_mean": 0.010253431391902268, "clip_ratio/low_min": 0.010253431391902268, "clip_ratio/region_mean": 0.015347328153438866, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 72.25, "completions/mean_terminated_length": 72.25, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.09350762516260147, "epoch": 0.03570711330682411, "frac_reward_zero_std": 0.0, "grad_norm": 3.3663084506988525, "learning_rate": 7.30909090909091e-06, "loss": 0.0078, "num_tokens": 1987943.0, "reward": 0.9931055307388306, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9931055307388306, "reward_meter_std": 0.0025689646136015654, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002568964147940278, "reward_total_composite_mean": 0.9931055307388306, "reward_total_composite_std": 0.0025689646136015654, "reward_total_mean": 0.9931055307388306, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9931055307388306, "rewards/meter/std": 0.0025689646136015654, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931055307388306, "rewards/total_composite/std": 0.0025689646136015654, "sampling/importance_sampling_ratio/max": 1.9120845794677734, "sampling/importance_sampling_ratio/mean": 1.0010665655136108, "sampling/importance_sampling_ratio/min": 0.2618023157119751, "sampling/sampling_logp_difference/max": 1.340165615081787, "sampling/sampling_logp_difference/mean": 0.021964317187666893, "step": 889 }, { "clip_ratio/high_max": 0.002556844614446163, "clip_ratio/high_mean": 0.002556844614446163, "clip_ratio/low_mean": 0.014987432223279029, "clip_ratio/low_min": 0.014987432223279029, "clip_ratio/region_mean": 0.017544276837725192, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 145.375, "completions/mean_terminated_length": 145.375, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.146129728294909, "epoch": 0.03574727878860907, "frac_reward_zero_std": 0.0, "grad_norm": 2.6016640663146973, "learning_rate": 7.306060606060607e-06, "loss": -0.005, "num_tokens": 1990330.0, "reward": 0.7374788522720337, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9830442667007446, "reward_meter_std": 0.01036095805466175, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06814645975828171, "reward_total_composite_mean": 0.7374788522720337, "reward_total_composite_std": 0.0681464672088623, "reward_total_mean": 0.7374788522720337, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9830442667007446, "rewards/meter/std": 0.01036095805466175, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7374788522720337, "rewards/total_composite/std": 0.0681464672088623, "sampling/importance_sampling_ratio/max": 1.7661346197128296, "sampling/importance_sampling_ratio/mean": 1.0078076124191284, "sampling/importance_sampling_ratio/min": 0.361492782831192, "sampling/sampling_logp_difference/max": 1.0175132751464844, "sampling/sampling_logp_difference/mean": 0.021481409668922424, "step": 890 }, { "clip_ratio/high_max": 0.018530045170336962, "clip_ratio/high_mean": 0.018530045170336962, "clip_ratio/low_mean": 0.0016666667070239782, "clip_ratio/low_min": 0.0016666667070239782, "clip_ratio/region_mean": 0.02019671187736094, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.19349019043147564, "epoch": 0.03578744427039402, "frac_reward_zero_std": 0.0, "grad_norm": 2.81315279006958, "learning_rate": 7.303030303030304e-06, "loss": 0.0093, "num_tokens": 1992094.0, "reward": 0.8690831661224365, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8690831661224365, "reward_meter_std": 0.3090604245662689, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.30906039476394653, "reward_total_composite_mean": 0.8690831661224365, "reward_total_composite_std": 0.3090604245662689, "reward_total_mean": 0.8690831661224365, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8690831661224365, "rewards/meter/std": 0.3090604245662689, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8690831661224365, "rewards/total_composite/std": 0.3090604245662689, "sampling/importance_sampling_ratio/max": 1.6693646907806396, "sampling/importance_sampling_ratio/mean": 1.0030601024627686, "sampling/importance_sampling_ratio/min": 0.24734534323215485, "sampling/sampling_logp_difference/max": 1.3969697952270508, "sampling/sampling_logp_difference/mean": 0.03247151896357536, "step": 891 }, { "clip_ratio/high_max": 0.003402777831070125, "clip_ratio/high_mean": 0.003402777831070125, "clip_ratio/low_mean": 0.013621801743283868, "clip_ratio/low_min": 0.013621801743283868, "clip_ratio/region_mean": 0.017024579574353993, "completions/clipped_ratio": 0.0, "completions/max_length": 115.0, "completions/max_terminated_length": 115.0, "completions/mean_length": 99.625, "completions/mean_terminated_length": 99.625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.1309506855905056, "epoch": 0.035827609752178975, "frac_reward_zero_std": 0.0, "grad_norm": 4.598532199859619, "learning_rate": 7.3e-06, "loss": 0.1432, "num_tokens": 1994275.0, "reward": 0.5349372625350952, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.2314550280570984, "reward_meter_mean": 0.994989275932312, "reward_meter_std": 0.0016277554677799344, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.12817399203777313, "reward_std": 0.2865643799304962, "reward_total_composite_mean": 0.5349372625350952, "reward_total_composite_std": 0.2865643799304962, "reward_total_mean": 0.5349372625350952, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.2314550280570984, "rewards/meter/mean": 0.994989275932312, "rewards/meter/std": 0.0016277554677799344, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.5349372625350952, "rewards/total_composite/std": 0.2865643799304962, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0079816579818726, "sampling/importance_sampling_ratio/min": 0.14815859496593475, "sampling/sampling_logp_difference/max": 1.9094719886779785, "sampling/sampling_logp_difference/mean": 0.02673969604074955, "step": 892 }, { "clip_ratio/high_max": 0.009196062572300434, "clip_ratio/high_mean": 0.009196062572300434, "clip_ratio/low_mean": 0.006828531681094319, "clip_ratio/low_min": 0.006828531681094319, "clip_ratio/region_mean": 0.016024594253394753, "completions/clipped_ratio": 0.0, "completions/max_length": 151.0, "completions/max_terminated_length": 151.0, "completions/mean_length": 146.25, "completions/mean_terminated_length": 146.25, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.13604794908314943, "epoch": 0.03586777523396393, "frac_reward_zero_std": 0.0, "grad_norm": 3.3807826042175293, "learning_rate": 7.296969696969698e-06, "loss": -0.0066, "num_tokens": 1996973.0, "reward": 0.017994364723563194, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.024410676211118698, "reward_meter_std": 0.046276357024908066, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.033032339066267014, "reward_total_composite_mean": 0.017994364723563194, "reward_total_composite_std": 0.03303234279155731, "reward_total_mean": 0.017994364723563194, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.024410676211118698, "rewards/meter/std": 0.046276357024908066, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.017994364723563194, "rewards/total_composite/std": 0.03303234279155731, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000627875328064, "sampling/importance_sampling_ratio/min": 0.17563189566135406, "sampling/sampling_logp_difference/max": 1.7393649816513062, "sampling/sampling_logp_difference/mean": 0.023801125586032867, "step": 893 }, { "clip_ratio/high_max": 0.005890377098694444, "clip_ratio/high_mean": 0.005890377098694444, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005890377098694444, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.625, "completions/mean_terminated_length": 64.625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.04335960140451789, "epoch": 0.03590794071574888, "frac_reward_zero_std": 0.0, "grad_norm": 3.678330659866333, "learning_rate": 7.293939393939394e-06, "loss": 0.0158, "num_tokens": 1998826.0, "reward": 0.957672119140625, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.957672119140625, "reward_meter_std": 0.010208838619291782, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01020884234458208, "reward_total_composite_mean": 0.957672119140625, "reward_total_composite_std": 0.010208838619291782, "reward_total_mean": 0.957672119140625, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.957672119140625, "rewards/meter/std": 0.010208838619291782, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.957672119140625, "rewards/total_composite/std": 0.010208838619291782, "sampling/importance_sampling_ratio/max": 1.4367311000823975, "sampling/importance_sampling_ratio/mean": 0.9999814033508301, "sampling/importance_sampling_ratio/min": 0.2768797278404236, "sampling/sampling_logp_difference/max": 1.2841720581054688, "sampling/sampling_logp_difference/mean": 0.00946664996445179, "step": 894 }, { "clip_ratio/high_max": 0.012951546465046704, "clip_ratio/high_mean": 0.012951546465046704, "clip_ratio/low_mean": 0.011278833262622356, "clip_ratio/low_min": 0.011278833262622356, "clip_ratio/region_mean": 0.02423037972766906, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.07575948582962155, "epoch": 0.03594810619753384, "frac_reward_zero_std": 0.0, "grad_norm": 3.20367169380188, "learning_rate": 7.290909090909092e-06, "loss": -0.0048, "num_tokens": 2000706.0, "reward": 0.6329219341278076, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6329219341278076, "reward_meter_std": 0.4643419682979584, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4643419682979584, "reward_total_composite_mean": 0.6329219341278076, "reward_total_composite_std": 0.4643419682979584, "reward_total_mean": 0.6329219341278076, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6329219341278076, "rewards/meter/std": 0.4643419682979584, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6329219341278076, "rewards/total_composite/std": 0.4643419682979584, "sampling/importance_sampling_ratio/max": 1.8657705783843994, "sampling/importance_sampling_ratio/mean": 1.006116509437561, "sampling/importance_sampling_ratio/min": 0.5214443206787109, "sampling/sampling_logp_difference/max": 0.6511527895927429, "sampling/sampling_logp_difference/mean": 0.015903156250715256, "step": 895 }, { "clip_ratio/high_max": 0.03242240101099014, "clip_ratio/high_mean": 0.03242240101099014, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/region_mean": 0.03999815881252289, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 63.125, "completions/mean_terminated_length": 63.125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.24534680880606174, "epoch": 0.03598827167931879, "frac_reward_zero_std": 0.0, "grad_norm": 17.783985137939453, "learning_rate": 7.287878787878789e-06, "loss": 0.0424, "num_tokens": 2002459.0, "reward": 0.926885187625885, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.926885187625885, "reward_meter_std": 0.12749086320400238, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.12749086320400238, "reward_total_composite_mean": 0.926885187625885, "reward_total_composite_std": 0.12749086320400238, "reward_total_mean": 0.926885187625885, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.926885187625885, "rewards/meter/std": 0.12749086320400238, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.926885187625885, "rewards/total_composite/std": 0.12749086320400238, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9990435838699341, "sampling/importance_sampling_ratio/min": 0.20310382544994354, "sampling/sampling_logp_difference/max": 1.5940380096435547, "sampling/sampling_logp_difference/mean": 0.04450727999210358, "step": 896 }, { "clip_ratio/high_max": 0.02354571979958564, "clip_ratio/high_mean": 0.02354571979958564, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/region_mean": 0.0281753494637087, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 75.75, "completions/mean_terminated_length": 75.75, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.3472878560423851, "epoch": 0.036028437161103745, "frac_reward_zero_std": 0.0, "grad_norm": 9.401634216308594, "learning_rate": 7.284848484848486e-06, "loss": 0.0518, "num_tokens": 2004481.0, "reward": 0.9812333583831787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9812333583831787, "reward_meter_std": 0.03400373458862305, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03400372713804245, "reward_total_composite_mean": 0.9812333583831787, "reward_total_composite_std": 0.03400373458862305, "reward_total_mean": 0.9812333583831787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9812333583831787, "rewards/meter/std": 0.03400373458862305, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9812333583831787, "rewards/total_composite/std": 0.03400373458862305, "sampling/importance_sampling_ratio/max": 1.990276575088501, "sampling/importance_sampling_ratio/mean": 1.0046457052230835, "sampling/importance_sampling_ratio/min": 0.31145650148391724, "sampling/sampling_logp_difference/max": 1.1664955615997314, "sampling/sampling_logp_difference/mean": 0.04621249809861183, "step": 897 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 64.125, "completions/mean_terminated_length": 64.125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.07330859638750553, "epoch": 0.0360686026428887, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.281818181818182e-06, "loss": 0.0, "num_tokens": 2006258.0, "reward": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9789108037948608, "reward_meter_std": 0.013038679957389832, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "reward_total_mean": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9789108037948608, "rewards/meter/std": 0.013038679957389832, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.383298635482788, "sampling/importance_sampling_ratio/mean": 1.0009260177612305, "sampling/importance_sampling_ratio/min": 0.2404828518629074, "sampling/sampling_logp_difference/max": 1.4251065254211426, "sampling/sampling_logp_difference/mean": 0.01310182549059391, "step": 898 }, { "clip_ratio/high_max": 0.010862766648642719, "clip_ratio/high_mean": 0.010862766648642719, "clip_ratio/low_mean": 0.004611999727785587, "clip_ratio/low_min": 0.004611999727785587, "clip_ratio/region_mean": 0.015474766376428306, "completions/clipped_ratio": 0.0, "completions/max_length": 117.0, "completions/max_terminated_length": 117.0, "completions/mean_length": 111.25, "completions/mean_terminated_length": 111.25, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.1524664619937539, "epoch": 0.03610876812467365, "frac_reward_zero_std": 0.0, "grad_norm": 3.2133944034576416, "learning_rate": 7.2787878787878795e-06, "loss": -0.0124, "num_tokens": 2008460.0, "reward": 0.6480754613876343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7485231161117554, "reward_meter_std": 0.4493827819824219, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.3985969126224518, "reward_total_composite_mean": 0.6480754613876343, "reward_total_composite_std": 0.3985969126224518, "reward_total_mean": 0.6480754613876343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7485231161117554, "rewards/meter/std": 0.4493827819824219, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6480754613876343, "rewards/total_composite/std": 0.3985969126224518, "sampling/importance_sampling_ratio/max": 1.8294382095336914, "sampling/importance_sampling_ratio/mean": 1.003222942352295, "sampling/importance_sampling_ratio/min": 0.36747586727142334, "sampling/sampling_logp_difference/max": 1.0010976791381836, "sampling/sampling_logp_difference/mean": 0.021135006099939346, "step": 899 }, { "clip_ratio/high_max": 0.02068706788122654, "clip_ratio/high_mean": 0.02068706788122654, "clip_ratio/low_mean": 0.024088865146040916, "clip_ratio/low_min": 0.024088865146040916, "clip_ratio/region_mean": 0.044775933027267456, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.4540216512978077, "epoch": 0.03614893360645861, "frac_reward_zero_std": 0.0, "grad_norm": 6.62265157699585, "learning_rate": 7.275757575757576e-06, "loss": 0.0111, "num_tokens": 2010132.0, "reward": 0.47188466787338257, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.47188466787338257, "reward_meter_std": 0.3173547685146332, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3173547685146332, "reward_total_composite_mean": 0.47188466787338257, "reward_total_composite_std": 0.3173547685146332, "reward_total_mean": 0.47188466787338257, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.47188466787338257, "rewards/meter/std": 0.3173547685146332, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.47188466787338257, "rewards/total_composite/std": 0.3173547685146332, "sampling/importance_sampling_ratio/max": 1.9393072128295898, "sampling/importance_sampling_ratio/mean": 1.0136620998382568, "sampling/importance_sampling_ratio/min": 0.34882447123527527, "sampling/sampling_logp_difference/max": 1.0531864166259766, "sampling/sampling_logp_difference/mean": 0.05849804729223251, "step": 900 }, { "epoch": 0.03614893360645861, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/max_length": 402.61538461538464, "eval_completions/max_terminated_length": 370.38461538461536, "eval_completions/mean_length": 220.43269230769232, "eval_completions/mean_terminated_length": 212.01373877892127, "eval_completions/min_length": 64.38461538461539, "eval_completions/min_terminated_length": 64.38461538461539, "eval_entropy": 0.0762252899316641, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 2010132.0, "eval_reward": 0.3643972415190477, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.8996222019195557, "eval_reward_count_adherence_std": 0.13804738968610764, "eval_reward_meter_mean": 0.607175561097952, "eval_reward_meter_std": 0.4337661495575538, "eval_reward_repeat_penalty_mean": 0.6879472640844492, "eval_reward_repeat_penalty_std": 0.20650707471829194, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.3643972415190477, "eval_reward_total_composite_std": 0.31932372657152325, "eval_reward_total_mean": 0.3643972415190477, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.8996222019195557, "eval_rewards/count_adherence/std": 0.13804738968610764, "eval_rewards/meter/mean": 0.607175561097952, "eval_rewards/meter/std": 0.4337661495575538, "eval_rewards/repeat_penalty/mean": 0.6879472640844492, "eval_rewards/repeat_penalty/std": 0.20650707471829194, "eval_rewards/total_composite/mean": 0.3643972415190477, "eval_rewards/total_composite/std": 0.31932372657152325, "eval_runtime": 76.3688, "eval_samples_per_second": 1.362, "eval_sampling/importance_sampling_ratio/max": 1.312718914105342, "eval_sampling/importance_sampling_ratio/mean": 1.002149930367103, "eval_sampling/importance_sampling_ratio/min": 0.4500999932105725, "eval_sampling/sampling_logp_difference/max": 0.8173254269819993, "eval_sampling/sampling_logp_difference/mean": 0.008556847639668446, "eval_steps_per_second": 0.17, "step": 900 }, { "clip_ratio/high_max": 0.00974025996401906, "clip_ratio/high_mean": 0.00974025996401906, "clip_ratio/low_mean": 0.02774310251697898, "clip_ratio/low_min": 0.02774310251697898, "clip_ratio/region_mean": 0.03748336248099804, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 65.25, "completions/mean_terminated_length": 65.25, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.39315336383879185, "epoch": 0.03618909908824356, "frac_reward_zero_std": 0.0, "grad_norm": 4.90191650390625, "learning_rate": 7.272727272727273e-06, "loss": 0.0221, "num_tokens": 2011886.0, "reward": 0.2965307831764221, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2965307831764221, "reward_meter_std": 0.3661009669303894, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.366100937128067, "reward_total_composite_mean": 0.2965307831764221, "reward_total_composite_std": 0.3661009669303894, "reward_total_mean": 0.2965307831764221, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2965307831764221, "rewards/meter/std": 0.3661009669303894, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2965307831764221, "rewards/total_composite/std": 0.3661009669303894, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0055922269821167, "sampling/importance_sampling_ratio/min": 0.28164607286453247, "sampling/sampling_logp_difference/max": 1.267104148864746, "sampling/sampling_logp_difference/mean": 0.05390980467200279, "step": 901 }, { "clip_ratio/high_max": 0.016934875398874283, "clip_ratio/high_mean": 0.016934875398874283, "clip_ratio/low_mean": 0.011442428571172059, "clip_ratio/low_min": 0.011442428571172059, "clip_ratio/region_mean": 0.02837730397004634, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 108.0, "completions/mean_terminated_length": 108.0, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.10153948981314898, "epoch": 0.036229264570028515, "frac_reward_zero_std": 0.0, "grad_norm": 2.6275370121002197, "learning_rate": 7.26969696969697e-06, "loss": 0.1103, "num_tokens": 2014158.0, "reward": 0.6347411870956421, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333730697632, "reward_count_adherence_std": 0.17817415297031403, "reward_meter_mean": 0.9951065182685852, "reward_meter_std": 0.0021594627760350704, "reward_repeat_penalty_mean": 0.7571429014205933, "reward_repeat_penalty_std": 0.04581620916724205, "reward_std": 0.17126502096652985, "reward_total_composite_mean": 0.6347411870956421, "reward_total_composite_std": 0.17126502096652985, "reward_total_mean": 0.6347411870956421, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333730697632, "rewards/count_adherence/std": 0.17817415297031403, "rewards/meter/mean": 0.9951065182685852, "rewards/meter/std": 0.0021594627760350704, "rewards/repeat_penalty/mean": 0.7571429014205933, "rewards/repeat_penalty/std": 0.04581620916724205, "rewards/total_composite/mean": 0.6347411870956421, "rewards/total_composite/std": 0.17126502096652985, "sampling/importance_sampling_ratio/max": 1.7781134843826294, "sampling/importance_sampling_ratio/mean": 1.0035145282745361, "sampling/importance_sampling_ratio/min": 0.37451860308647156, "sampling/sampling_logp_difference/max": 0.9821138381958008, "sampling/sampling_logp_difference/mean": 0.019786102697253227, "step": 902 }, { "clip_ratio/high_max": 0.009641934651881456, "clip_ratio/high_mean": 0.009641934651881456, "clip_ratio/low_mean": 0.0012135922443121672, "clip_ratio/low_min": 0.0012135922443121672, "clip_ratio/region_mean": 0.010855526896193624, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 103.75, "completions/mean_terminated_length": 103.75, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.05680669890716672, "epoch": 0.03626943005181347, "frac_reward_zero_std": 0.0, "grad_norm": 1.540696620941162, "learning_rate": 7.266666666666668e-06, "loss": -0.0025, "num_tokens": 2016340.0, "reward": 0.773260235786438, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977596998214722, "reward_meter_std": 0.00013270843192003667, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07051652669906616, "reward_total_composite_mean": 0.773260235786438, "reward_total_composite_std": 0.07051651179790497, "reward_total_mean": 0.773260235786438, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977596998214722, "rewards/meter/std": 0.00013270843192003667, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.773260235786438, "rewards/total_composite/std": 0.07051651179790497, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0004560947418213, "sampling/importance_sampling_ratio/min": 0.32154497504234314, "sampling/sampling_logp_difference/max": 1.134617805480957, "sampling/sampling_logp_difference/mean": 0.012240896932780743, "step": 903 }, { "clip_ratio/high_max": 0.02894315170124173, "clip_ratio/high_mean": 0.02894315170124173, "clip_ratio/low_mean": 0.00657894741743803, "clip_ratio/low_min": 0.00657894741743803, "clip_ratio/region_mean": 0.03552209911867976, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 97.125, "completions/mean_terminated_length": 37.85714340209961, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.5050923340022564, "epoch": 0.03630959553359842, "frac_reward_zero_std": 0.0, "grad_norm": 2.963177442550659, "learning_rate": 7.263636363636364e-06, "loss": -0.0841, "num_tokens": 2017997.0, "reward": 0.7298738360404968, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8145281076431274, "reward_meter_std": 0.27075159549713135, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3964882493019104, "reward_total_composite_mean": 0.7298738360404968, "reward_total_composite_std": 0.3964882493019104, "reward_total_mean": 0.7298738360404968, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8145281076431274, "rewards/meter/std": 0.27075159549713135, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7298738360404968, "rewards/total_composite/std": 0.3964882493019104, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0168981552124023, "sampling/importance_sampling_ratio/min": 0.1524311900138855, "sampling/sampling_logp_difference/max": 1.8810420036315918, "sampling/sampling_logp_difference/mean": 0.09322468191385269, "step": 904 }, { "clip_ratio/high_max": 0.004120342840906233, "clip_ratio/high_mean": 0.004120342840906233, "clip_ratio/low_mean": 0.00478401588043198, "clip_ratio/low_min": 0.00478401588043198, "clip_ratio/region_mean": 0.008904358721338212, "completions/clipped_ratio": 0.0, "completions/max_length": 162.0, "completions/max_terminated_length": 162.0, "completions/mean_length": 153.25, "completions/mean_terminated_length": 153.25, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.048865064047276974, "epoch": 0.036349761015383376, "frac_reward_zero_std": 0.0, "grad_norm": 1.4031760692596436, "learning_rate": 7.260606060606061e-06, "loss": 0.0081, "num_tokens": 2020527.0, "reward": 0.5992619395256042, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9857839941978455, "reward_meter_std": 0.010634319856762886, "reward_repeat_penalty_mean": 0.6071429252624512, "reward_repeat_penalty_std": 0.147871196269989, "reward_std": 0.14893361926078796, "reward_total_composite_mean": 0.5992619395256042, "reward_total_composite_std": 0.14893361926078796, "reward_total_mean": 0.5992619395256042, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9857839941978455, "rewards/meter/std": 0.010634319856762886, "rewards/repeat_penalty/mean": 0.6071429252624512, "rewards/repeat_penalty/std": 0.147871196269989, "rewards/total_composite/mean": 0.5992619395256042, "rewards/total_composite/std": 0.14893361926078796, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0026439428329468, "sampling/importance_sampling_ratio/min": 0.5704922080039978, "sampling/sampling_logp_difference/max": 0.697784423828125, "sampling/sampling_logp_difference/mean": 0.008156189695000648, "step": 905 }, { "clip_ratio/high_max": 0.03257575840689242, "clip_ratio/high_mean": 0.03257575840689242, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/region_mean": 0.036421912256628275, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 38.85714340209961, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.1836453713476658, "epoch": 0.03638992649716833, "frac_reward_zero_std": 0.0, "grad_norm": 3.238330125808716, "learning_rate": 7.257575757575758e-06, "loss": -0.0066, "num_tokens": 2022039.0, "reward": 0.7457597255706787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_meter_mean": 0.8625515699386597, "reward_meter_std": 0.34917646646499634, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4603087604045868, "reward_total_composite_mean": 0.7457597255706787, "reward_total_composite_std": 0.4603087902069092, "reward_total_mean": 0.7457597255706787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/meter/mean": 0.8625515699386597, "rewards/meter/std": 0.34917646646499634, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7457597255706787, "rewards/total_composite/std": 0.4603087902069092, "sampling/importance_sampling_ratio/max": 1.4525083303451538, "sampling/importance_sampling_ratio/mean": 0.988756537437439, "sampling/importance_sampling_ratio/min": 0.07324519753456116, "sampling/sampling_logp_difference/max": 2.6139426231384277, "sampling/sampling_logp_difference/mean": 0.053011078387498856, "step": 906 }, { "clip_ratio/high_max": 0.0027064846362918615, "clip_ratio/high_mean": 0.0027064846362918615, "clip_ratio/low_mean": 0.0029962139669805765, "clip_ratio/low_min": 0.0029962139669805765, "clip_ratio/region_mean": 0.005702698603272438, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 257.375, "completions/mean_terminated_length": 257.375, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "entropy": 0.030403183307498693, "epoch": 0.036430091978953284, "frac_reward_zero_std": 0.0, "grad_norm": 2.8838396072387695, "learning_rate": 7.254545454545455e-06, "loss": -0.0466, "num_tokens": 2025618.0, "reward": 0.058102987706661224, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9166666269302368, "reward_count_adherence_std": 0.05143444612622261, "reward_meter_mean": 0.1675853729248047, "reward_meter_std": 0.09037812799215317, "reward_repeat_penalty_mean": 0.31433823704719543, "reward_repeat_penalty_std": 0.18327650427818298, "reward_std": 0.0715017318725586, "reward_total_composite_mean": 0.058102987706661224, "reward_total_composite_std": 0.07150173932313919, "reward_total_mean": 0.058102987706661224, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9166666269302368, "rewards/count_adherence/std": 0.05143444612622261, "rewards/meter/mean": 0.1675853729248047, "rewards/meter/std": 0.09037812799215317, "rewards/repeat_penalty/mean": 0.31433823704719543, "rewards/repeat_penalty/std": 0.18327650427818298, "rewards/total_composite/mean": 0.058102987706661224, "rewards/total_composite/std": 0.07150173932313919, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0021930932998657, "sampling/importance_sampling_ratio/min": 0.2157527506351471, "sampling/sampling_logp_difference/max": 1.5336222648620605, "sampling/sampling_logp_difference/mean": 0.008167757652699947, "step": 907 }, { "clip_ratio/high_max": 0.007976973778568208, "clip_ratio/high_mean": 0.007976973778568208, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007976973778568208, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 132.0, "completions/mean_terminated_length": 77.71428680419922, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.07274728454649448, "epoch": 0.03647025746073824, "frac_reward_zero_std": 0.0, "grad_norm": 0.40971025824546814, "learning_rate": 7.251515151515151e-06, "loss": -0.183, "num_tokens": 2027442.0, "reward": 0.8669998645782471, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8670340180397034, "reward_meter_std": 0.3502245545387268, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35032105445861816, "reward_total_composite_mean": 0.8669998645782471, "reward_total_composite_std": 0.35032105445861816, "reward_total_mean": 0.8669998645782471, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8670340180397034, "rewards/meter/std": 0.3502245545387268, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8669998645782471, "rewards/total_composite/std": 0.35032105445861816, "sampling/importance_sampling_ratio/max": 1.2967445850372314, "sampling/importance_sampling_ratio/mean": 1.0027331113815308, "sampling/importance_sampling_ratio/min": 0.5214623212814331, "sampling/sampling_logp_difference/max": 0.651118278503418, "sampling/sampling_logp_difference/mean": 0.010826216079294682, "step": 908 }, { "clip_ratio/high_max": 0.002550398523453623, "clip_ratio/high_mean": 0.002550398523453623, "clip_ratio/low_mean": 0.003980891779065132, "clip_ratio/low_min": 0.003980891779065132, "clip_ratio/region_mean": 0.006531290302518755, "completions/clipped_ratio": 0.0, "completions/max_length": 158.0, "completions/max_terminated_length": 158.0, "completions/mean_length": 151.125, "completions/mean_terminated_length": 151.125, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.051733002765104175, "epoch": 0.03651042294252319, "frac_reward_zero_std": 0.0, "grad_norm": 2.3503267765045166, "learning_rate": 7.2484848484848495e-06, "loss": 0.0147, "num_tokens": 2030075.0, "reward": 0.6414377689361572, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8722712993621826, "reward_meter_std": 0.33660420775413513, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.2504880726337433, "reward_total_composite_mean": 0.6414377689361572, "reward_total_composite_std": 0.2504880726337433, "reward_total_mean": 0.6414377689361572, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8722712993621826, "rewards/meter/std": 0.33660420775413513, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.6414377689361572, "rewards/total_composite/std": 0.2504880726337433, "sampling/importance_sampling_ratio/max": 1.8332045078277588, "sampling/importance_sampling_ratio/mean": 1.0012683868408203, "sampling/importance_sampling_ratio/min": 0.395646870136261, "sampling/sampling_logp_difference/max": 0.9272332191467285, "sampling/sampling_logp_difference/mean": 0.009546966291964054, "step": 909 }, { "clip_ratio/high_max": 0.005175159312784672, "clip_ratio/high_mean": 0.005175159312784672, "clip_ratio/low_mean": 0.011030830384697765, "clip_ratio/low_min": 0.011030830384697765, "clip_ratio/region_mean": 0.016205989697482437, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 297.0, "completions/mean_terminated_length": 297.0, "completions/min_length": 274.0, "completions/min_terminated_length": 274.0, "entropy": 0.10688875103369355, "epoch": 0.03655058842430815, "frac_reward_zero_std": 0.0, "grad_norm": 1.5391802787780762, "learning_rate": 7.245454545454546e-06, "loss": -0.0173, "num_tokens": 2034379.0, "reward": 0.06387612223625183, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.08010874688625336, "reward_meter_mean": 0.10647714138031006, "reward_meter_std": 0.29925623536109924, "reward_repeat_penalty_mean": 0.6176573634147644, "reward_repeat_penalty_std": 0.15113945305347443, "reward_std": 0.1795579344034195, "reward_total_composite_mean": 0.06387612223625183, "reward_total_composite_std": 0.1795579344034195, "reward_total_mean": 0.06387612223625183, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.08010874688625336, "rewards/meter/mean": 0.10647714138031006, "rewards/meter/std": 0.29925623536109924, "rewards/repeat_penalty/mean": 0.6176573634147644, "rewards/repeat_penalty/std": 0.15113945305347443, "rewards/total_composite/mean": 0.06387612223625183, "rewards/total_composite/std": 0.1795579344034195, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.999440610408783, "sampling/importance_sampling_ratio/min": 0.0067737954668700695, "sampling/sampling_logp_difference/max": 4.994693756103516, "sampling/sampling_logp_difference/mean": 0.02341293916106224, "step": 910 }, { "clip_ratio/high_max": 0.003086419776082039, "clip_ratio/high_mean": 0.003086419776082039, "clip_ratio/low_mean": 0.012957202387042344, "clip_ratio/low_min": 0.012957202387042344, "clip_ratio/region_mean": 0.016043622163124382, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 130.5, "completions/mean_terminated_length": 76.0, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.11146194022148848, "epoch": 0.03659075390609311, "frac_reward_zero_std": 0.0, "grad_norm": 2.3471314907073975, "learning_rate": 7.242424242424243e-06, "loss": -0.0319, "num_tokens": 2036087.0, "reward": 0.0003843327867798507, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.00040913556586019695, "reward_meter_std": 0.00048293921281583607, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004887212999165058, "reward_total_composite_mean": 0.0003843327867798507, "reward_total_composite_std": 0.0004887212999165058, "reward_total_mean": 0.0003843327867798507, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.00040913556586019695, "rewards/meter/std": 0.00048293921281583607, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0003843327867798507, "rewards/total_composite/std": 0.0004887212999165058, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9987106919288635, "sampling/importance_sampling_ratio/min": 0.2339477390050888, "sampling/sampling_logp_difference/max": 1.4526575803756714, "sampling/sampling_logp_difference/mean": 0.03155261650681496, "step": 911 }, { "clip_ratio/high_max": 0.002673796727322042, "clip_ratio/high_mean": 0.002673796727322042, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.002673796727322042, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 187.0, "completions/mean_length": 227.625, "completions/mean_terminated_length": 187.00001525878906, "completions/min_length": 187.0, "completions/min_terminated_length": 187.0, "entropy": 0.026463798945769668, "epoch": 0.03663091938787806, "frac_reward_zero_std": 0.0, "grad_norm": 0.514905571937561, "learning_rate": 7.2393939393939404e-06, "loss": -0.2509, "num_tokens": 2038908.0, "reward": 0.5911375284194946, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.8691276907920837, "reward_meter_std": 0.3413867950439453, "reward_repeat_penalty_mean": 0.7222222089767456, "reward_repeat_penalty_std": 0.11878277361392975, "reward_std": 0.24190570414066315, "reward_total_composite_mean": 0.5911375284194946, "reward_total_composite_std": 0.24190568923950195, "reward_total_mean": 0.5911375284194946, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.8691276907920837, "rewards/meter/std": 0.3413867950439453, "rewards/repeat_penalty/mean": 0.7222222089767456, "rewards/repeat_penalty/std": 0.11878277361392975, "rewards/total_composite/mean": 0.5911375284194946, "rewards/total_composite/std": 0.24190568923950195, "sampling/importance_sampling_ratio/max": 1.1373112201690674, "sampling/importance_sampling_ratio/mean": 1.0010285377502441, "sampling/importance_sampling_ratio/min": 0.42816048860549927, "sampling/sampling_logp_difference/max": 0.8482571840286255, "sampling/sampling_logp_difference/mean": 0.0038684571627527475, "step": 912 }, { "clip_ratio/high_max": 0.0011353294248692691, "clip_ratio/high_mean": 0.0011353294248692691, "clip_ratio/low_mean": 0.004096297547221184, "clip_ratio/low_min": 0.004096297547221184, "clip_ratio/region_mean": 0.005231626972090453, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 378.5, "completions/mean_terminated_length": 334.0, "completions/min_length": 328.0, "completions/min_terminated_length": 328.0, "entropy": 0.04513176158070564, "epoch": 0.036671084869663015, "frac_reward_zero_std": 0.0, "grad_norm": 1.2060858011245728, "learning_rate": 7.236363636363637e-06, "loss": -0.0935, "num_tokens": 2042728.0, "reward": 0.08265350759029388, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6477272510528564, "reward_count_adherence_std": 0.24021585285663605, "reward_meter_mean": 0.24957457184791565, "reward_meter_std": 0.42763951420783997, "reward_repeat_penalty_mean": 0.5911239385604858, "reward_repeat_penalty_std": 0.21299511194229126, "reward_std": 0.1630992740392685, "reward_total_composite_mean": 0.08265350759029388, "reward_total_composite_std": 0.1630992591381073, "reward_total_mean": 0.08265350759029388, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6477272510528564, "rewards/count_adherence/std": 0.24021585285663605, "rewards/meter/mean": 0.24957457184791565, "rewards/meter/std": 0.42763951420783997, "rewards/repeat_penalty/mean": 0.5911239385604858, "rewards/repeat_penalty/std": 0.21299511194229126, "rewards/total_composite/mean": 0.08265350759029388, "rewards/total_composite/std": 0.1630992591381073, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017143487930298, "sampling/importance_sampling_ratio/min": 0.11471810191869736, "sampling/sampling_logp_difference/max": 2.1652774810791016, "sampling/sampling_logp_difference/mean": 0.01384922955185175, "step": 913 }, { "clip_ratio/high_max": 0.009057971183210611, "clip_ratio/high_mean": 0.009057971183210611, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/region_mean": 0.014415114186704159, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.25, "completions/mean_terminated_length": 69.25, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.09249463677406311, "epoch": 0.03671125035144797, "frac_reward_zero_std": 0.0, "grad_norm": 3.0142698287963867, "learning_rate": 7.233333333333334e-06, "loss": 0.0031, "num_tokens": 2044450.0, "reward": 0.9978233575820923, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978233575820923, "reward_meter_std": 0.000473748950753361, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004737446433864534, "reward_total_composite_mean": 0.9978233575820923, "reward_total_composite_std": 0.000473748950753361, "reward_total_mean": 0.9978233575820923, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978233575820923, "rewards/meter/std": 0.000473748950753361, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978233575820923, "rewards/total_composite/std": 0.000473748950753361, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0037966966629028, "sampling/importance_sampling_ratio/min": 0.4556933641433716, "sampling/sampling_logp_difference/max": 0.8328394889831543, "sampling/sampling_logp_difference/mean": 0.017529845237731934, "step": 914 }, { "clip_ratio/high_max": 0.00370671006385237, "clip_ratio/high_mean": 0.00370671006385237, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.005600649514235556, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 410.25, "completions/mean_terminated_length": 308.5, "completions/min_length": 296.0, "completions/min_terminated_length": 296.0, "entropy": 0.03112065652385354, "epoch": 0.03675141583323292, "frac_reward_zero_std": 0.0, "grad_norm": 1.0550512075424194, "learning_rate": 7.2303030303030305e-06, "loss": -0.2438, "num_tokens": 2047404.0, "reward": 0.1930341273546219, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.4027777910232544, "reward_count_adherence_std": 0.36581191420555115, "reward_meter_mean": 0.7131782174110413, "reward_meter_std": 0.40048038959503174, "reward_repeat_penalty_mean": 0.8327265977859497, "reward_repeat_penalty_std": 0.20120510458946228, "reward_std": 0.20283441245555878, "reward_total_composite_mean": 0.1930341273546219, "reward_total_composite_std": 0.20283441245555878, "reward_total_mean": 0.1930341273546219, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.4027777910232544, "rewards/count_adherence/std": 0.36581191420555115, "rewards/meter/mean": 0.7131782174110413, "rewards/meter/std": 0.40048038959503174, "rewards/repeat_penalty/mean": 0.8327265977859497, "rewards/repeat_penalty/std": 0.20120510458946228, "rewards/total_composite/mean": 0.1930341273546219, "rewards/total_composite/std": 0.20283441245555878, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016696453094482, "sampling/importance_sampling_ratio/min": 0.12874329090118408, "sampling/sampling_logp_difference/max": 2.0499348640441895, "sampling/sampling_logp_difference/mean": 0.016473982483148575, "step": 915 }, { "clip_ratio/high_max": 0.021643032785505056, "clip_ratio/high_mean": 0.021643032785505056, "clip_ratio/low_mean": 0.02030206471681595, "clip_ratio/low_min": 0.02030206471681595, "clip_ratio/region_mean": 0.041945097502321005, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 54.25, "completions/mean_terminated_length": 54.25, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.36291532032191753, "epoch": 0.03679158131501788, "frac_reward_zero_std": 0.0, "grad_norm": 6.668519973754883, "learning_rate": 7.227272727272729e-06, "loss": 0.1148, "num_tokens": 2048990.0, "reward": 0.4157664477825165, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4157664477825165, "reward_meter_std": 0.3722110092639923, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3722110092639923, "reward_total_composite_mean": 0.4157664477825165, "reward_total_composite_std": 0.3722110092639923, "reward_total_mean": 0.4157664477825165, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4157664477825165, "rewards/meter/std": 0.3722110092639923, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4157664477825165, "rewards/total_composite/std": 0.3722110092639923, "sampling/importance_sampling_ratio/max": 1.8987327814102173, "sampling/importance_sampling_ratio/mean": 1.011838436126709, "sampling/importance_sampling_ratio/min": 0.12797655165195465, "sampling/sampling_logp_difference/max": 2.055908203125, "sampling/sampling_logp_difference/mean": 0.06838522106409073, "step": 916 }, { "clip_ratio/high_max": 0.009615384973585606, "clip_ratio/high_mean": 0.009615384973585606, "clip_ratio/low_mean": 0.05021517118439078, "clip_ratio/low_min": 0.05021517118439078, "clip_ratio/region_mean": 0.05983055615797639, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 81.375, "completions/mean_terminated_length": 81.375, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.2575049940496683, "epoch": 0.03683174679680283, "frac_reward_zero_std": 0.0, "grad_norm": 4.49209451675415, "learning_rate": 7.224242424242425e-06, "loss": 0.0255, "num_tokens": 2050857.0, "reward": 0.01962619088590145, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.024357754737138748, "reward_meter_std": 0.066401407122612, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.053068071603775024, "reward_total_composite_mean": 0.01962619088590145, "reward_total_composite_std": 0.053068071603775024, "reward_total_mean": 0.01962619088590145, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.024357754737138748, "rewards/meter/std": 0.066401407122612, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.01962619088590145, "rewards/total_composite/std": 0.053068071603775024, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002654790878296, "sampling/importance_sampling_ratio/min": 0.07217957079410553, "sampling/sampling_logp_difference/max": 2.628598213195801, "sampling/sampling_logp_difference/mean": 0.06128174066543579, "step": 917 }, { "clip_ratio/high_max": 0.01266416534781456, "clip_ratio/high_mean": 0.01266416534781456, "clip_ratio/low_mean": 0.015625000232830644, "clip_ratio/low_min": 0.015625000232830644, "clip_ratio/region_mean": 0.028289165580645204, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 40.0, "completions/mean_terminated_length": 40.0, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.1840200126171112, "epoch": 0.036871912278587785, "frac_reward_zero_std": 0.0, "grad_norm": 3.6781203746795654, "learning_rate": 7.221212121212122e-06, "loss": 0.0014, "num_tokens": 2052361.0, "reward": 0.25423121452331543, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.25423121452331543, "reward_meter_std": 0.42384836077690125, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.42384836077690125, "reward_total_composite_mean": 0.25423121452331543, "reward_total_composite_std": 0.42384836077690125, "reward_total_mean": 0.25423121452331543, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.25423121452331543, "rewards/meter/std": 0.42384836077690125, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.25423121452331543, "rewards/total_composite/std": 0.42384836077690125, "sampling/importance_sampling_ratio/max": 1.5022114515304565, "sampling/importance_sampling_ratio/mean": 1.0119025707244873, "sampling/importance_sampling_ratio/min": 0.5739105939865112, "sampling/sampling_logp_difference/max": 0.5552816390991211, "sampling/sampling_logp_difference/mean": 0.025724250823259354, "step": 918 }, { "clip_ratio/high_max": 0.023888979107141495, "clip_ratio/high_mean": 0.023888979107141495, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.027795229107141495, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.11905911657959223, "epoch": 0.03691207776037274, "frac_reward_zero_std": 0.0, "grad_norm": 3.6577181816101074, "learning_rate": 7.218181818181819e-06, "loss": -0.0107, "num_tokens": 2054233.0, "reward": 0.8576065301895142, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8576065301895142, "reward_meter_std": 0.346675843000412, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3466758131980896, "reward_total_composite_mean": 0.8576065301895142, "reward_total_composite_std": 0.346675843000412, "reward_total_mean": 0.8576065301895142, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8576065301895142, "rewards/meter/std": 0.346675843000412, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8576065301895142, "rewards/total_composite/std": 0.346675843000412, "sampling/importance_sampling_ratio/max": 1.7233705520629883, "sampling/importance_sampling_ratio/mean": 1.0019348859786987, "sampling/importance_sampling_ratio/min": 0.12677344679832458, "sampling/sampling_logp_difference/max": 2.0653536319732666, "sampling/sampling_logp_difference/mean": 0.024599751457571983, "step": 919 }, { "clip_ratio/high_max": 0.005704862414859235, "clip_ratio/high_mean": 0.005704862414859235, "clip_ratio/low_mean": 0.0014981299173086882, "clip_ratio/low_min": 0.0014981299173086882, "clip_ratio/region_mean": 0.0072029923321679235, "completions/clipped_ratio": 0.0, "completions/max_length": 334.0, "completions/max_terminated_length": 334.0, "completions/mean_length": 330.625, "completions/mean_terminated_length": 330.625, "completions/min_length": 313.0, "completions/min_terminated_length": 313.0, "entropy": 0.03318166173994541, "epoch": 0.03695224324215769, "frac_reward_zero_std": 0.0, "grad_norm": 0.8415653109550476, "learning_rate": 7.215151515151516e-06, "loss": 0.0055, "num_tokens": 2058494.0, "reward": 0.3763912320137024, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7692307829856873, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7762564420700073, "reward_meter_std": 0.19550828635692596, "reward_repeat_penalty_mean": 0.625, "reward_repeat_penalty_std": 0.0864252969622612, "reward_std": 0.11739426851272583, "reward_total_composite_mean": 0.3763912320137024, "reward_total_composite_std": 0.11739427596330643, "reward_total_mean": 0.3763912320137024, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7692307829856873, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7762564420700073, "rewards/meter/std": 0.19550828635692596, "rewards/repeat_penalty/mean": 0.625, "rewards/repeat_penalty/std": 0.0864252969622612, "rewards/total_composite/mean": 0.3763912320137024, "rewards/total_composite/std": 0.11739427596330643, "sampling/importance_sampling_ratio/max": 1.6070106029510498, "sampling/importance_sampling_ratio/mean": 0.9997749328613281, "sampling/importance_sampling_ratio/min": 0.39294660091400146, "sampling/sampling_logp_difference/max": 0.9340815544128418, "sampling/sampling_logp_difference/mean": 0.007077803369611502, "step": 920 }, { "clip_ratio/high_max": 0.029582271818071604, "clip_ratio/high_mean": 0.029582271818071604, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.029582271818071604, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 313.0, "completions/mean_terminated_length": 114.0, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.4247525855898857, "epoch": 0.036992408723942646, "frac_reward_zero_std": 0.0, "grad_norm": 1.4356052875518799, "learning_rate": 7.212121212121212e-06, "loss": -0.1723, "num_tokens": 2060102.0, "reward": 0.44831955432891846, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.33034372329711914, "reward_meter_mean": 0.842616856098175, "reward_meter_std": 0.3094305694103241, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.4836740791797638, "reward_total_composite_mean": 0.44831955432891846, "reward_total_composite_std": 0.4836740791797638, "reward_total_mean": 0.44831955432891846, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.33034372329711914, "rewards/meter/mean": 0.842616856098175, "rewards/meter/std": 0.3094305694103241, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.44831955432891846, "rewards/total_composite/std": 0.4836740791797638, "sampling/importance_sampling_ratio/max": 1.9944467544555664, "sampling/importance_sampling_ratio/mean": 1.0298396348953247, "sampling/importance_sampling_ratio/min": 0.4205147325992584, "sampling/sampling_logp_difference/max": 0.8662757873535156, "sampling/sampling_logp_difference/mean": 0.0745995044708252, "step": 921 }, { "clip_ratio/high_max": 0.001409774413332343, "clip_ratio/high_mean": 0.001409774413332343, "clip_ratio/low_mean": 0.012246111291460693, "clip_ratio/low_min": 0.012246111291460693, "clip_ratio/region_mean": 0.013655885704793036, "completions/clipped_ratio": 0.0, "completions/max_length": 266.0, "completions/max_terminated_length": 266.0, "completions/mean_length": 248.625, "completions/mean_terminated_length": 248.625, "completions/min_length": 238.0, "completions/min_terminated_length": 238.0, "entropy": 0.06539201783016324, "epoch": 0.0370325742057276, "frac_reward_zero_std": 0.0, "grad_norm": 1.704238772392273, "learning_rate": 7.2090909090909104e-06, "loss": -0.0231, "num_tokens": 2063555.0, "reward": 0.004727967549115419, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.141575887799263, "reward_meter_mean": 0.007957300171256065, "reward_meter_std": 0.0194723941385746, "reward_repeat_penalty_mean": 0.6661838293075562, "reward_repeat_penalty_std": 0.07680089771747589, "reward_std": 0.012029902078211308, "reward_total_composite_mean": 0.004727967549115419, "reward_total_composite_std": 0.012029902078211308, "reward_total_mean": 0.004727967549115419, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.141575887799263, "rewards/meter/mean": 0.007957300171256065, "rewards/meter/std": 0.0194723941385746, "rewards/repeat_penalty/mean": 0.6661838293075562, "rewards/repeat_penalty/std": 0.07680089771747589, "rewards/total_composite/mean": 0.004727967549115419, "rewards/total_composite/std": 0.012029902078211308, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9993036985397339, "sampling/importance_sampling_ratio/min": 0.06969470530748367, "sampling/sampling_logp_difference/max": 2.663630962371826, "sampling/sampling_logp_difference/mean": 0.02039029449224472, "step": 922 }, { "clip_ratio/high_max": 0.027541207149624825, "clip_ratio/high_mean": 0.027541207149624825, "clip_ratio/low_mean": 0.02332585956901312, "clip_ratio/low_min": 0.02332585956901312, "clip_ratio/region_mean": 0.05086706671863794, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 49.25, "completions/mean_terminated_length": 49.25, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.17652267403900623, "epoch": 0.037072739687512554, "frac_reward_zero_std": 0.0, "grad_norm": 10.85976791381836, "learning_rate": 7.206060606060606e-06, "loss": -0.059, "num_tokens": 2065157.0, "reward": 0.514995276927948, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.514995276927948, "reward_meter_std": 0.3742930591106415, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3742930591106415, "reward_total_composite_mean": 0.514995276927948, "reward_total_composite_std": 0.3742930591106415, "reward_total_mean": 0.514995276927948, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.514995276927948, "rewards/meter/std": 0.3742930591106415, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.514995276927948, "rewards/total_composite/std": 0.3742930591106415, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0013222694396973, "sampling/importance_sampling_ratio/min": 0.3031938970088959, "sampling/sampling_logp_difference/max": 1.193382740020752, "sampling/sampling_logp_difference/mean": 0.04450875520706177, "step": 923 }, { "clip_ratio/high_max": 0.019490185426548123, "clip_ratio/high_mean": 0.019490185426548123, "clip_ratio/low_mean": 0.01368181873112917, "clip_ratio/low_min": 0.01368181873112917, "clip_ratio/region_mean": 0.03317200415767729, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 214.875, "completions/mean_terminated_length": 115.83333587646484, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.3687305059283972, "epoch": 0.03711290516929751, "frac_reward_zero_std": 0.0, "grad_norm": 1.6212074756622314, "learning_rate": 7.203030303030304e-06, "loss": -0.127, "num_tokens": 2067180.0, "reward": 0.40130022168159485, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.594925582408905, "reward_meter_std": 0.46331310272216797, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.39831164479255676, "reward_total_composite_mean": 0.40130022168159485, "reward_total_composite_std": 0.3983116149902344, "reward_total_mean": 0.40130022168159485, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.594925582408905, "rewards/meter/std": 0.46331310272216797, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.40130022168159485, "rewards/total_composite/std": 0.3983116149902344, "sampling/importance_sampling_ratio/max": 1.917328119277954, "sampling/importance_sampling_ratio/mean": 1.0083558559417725, "sampling/importance_sampling_ratio/min": 0.3223177492618561, "sampling/sampling_logp_difference/max": 1.1322174072265625, "sampling/sampling_logp_difference/mean": 0.051524095237255096, "step": 924 }, { "clip_ratio/high_max": 0.03658821329008788, "clip_ratio/high_mean": 0.03658821329008788, "clip_ratio/low_mean": 0.010874067898839712, "clip_ratio/low_min": 0.010874067898839712, "clip_ratio/region_mean": 0.04746228118892759, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.36769964918494225, "epoch": 0.03715307065108246, "frac_reward_zero_std": 0.0, "grad_norm": 5.2026143074035645, "learning_rate": 7.2000000000000005e-06, "loss": -0.029, "num_tokens": 2069136.0, "reward": 0.7086162567138672, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8316957354545593, "reward_meter_std": 0.3097537159919739, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4172649383544922, "reward_total_composite_mean": 0.7086162567138672, "reward_total_composite_std": 0.4172649383544922, "reward_total_mean": 0.7086162567138672, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8316957354545593, "rewards/meter/std": 0.3097537159919739, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7086162567138672, "rewards/total_composite/std": 0.4172649383544922, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0118038654327393, "sampling/importance_sampling_ratio/min": 0.3303510844707489, "sampling/sampling_logp_difference/max": 1.1075992584228516, "sampling/sampling_logp_difference/mean": 0.051565829664468765, "step": 925 }, { "clip_ratio/high_max": 0.00898033136036247, "clip_ratio/high_mean": 0.00898033136036247, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.00898033136036247, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.5, "completions/mean_terminated_length": 69.5, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.06283332780003548, "epoch": 0.037193236132867416, "frac_reward_zero_std": 0.0, "grad_norm": 2.8455419540405273, "learning_rate": 7.196969696969698e-06, "loss": 0.001, "num_tokens": 2070956.0, "reward": 0.9984892010688782, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984892010688782, "reward_meter_std": 8.65640613483265e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.656938007334247e-05, "reward_total_composite_mean": 0.9984892010688782, "reward_total_composite_std": 8.65640613483265e-05, "reward_total_mean": 0.9984892010688782, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984892010688782, "rewards/meter/std": 8.65640613483265e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984892010688782, "rewards/total_composite/std": 8.65640613483265e-05, "sampling/importance_sampling_ratio/max": 1.7046222686767578, "sampling/importance_sampling_ratio/mean": 1.0004963874816895, "sampling/importance_sampling_ratio/min": 0.2656971514225006, "sampling/sampling_logp_difference/max": 1.3253982067108154, "sampling/sampling_logp_difference/mean": 0.015567532740533352, "step": 926 }, { "clip_ratio/high_max": 0.0010664711589924991, "clip_ratio/high_mean": 0.0010664711589924991, "clip_ratio/low_mean": 0.00428855000063777, "clip_ratio/low_min": 0.00428855000063777, "clip_ratio/region_mean": 0.005355021159630269, "completions/clipped_ratio": 0.0, "completions/max_length": 375.0, "completions/max_terminated_length": 375.0, "completions/mean_length": 347.5, "completions/mean_terminated_length": 347.5, "completions/min_length": 341.0, "completions/min_terminated_length": 341.0, "entropy": 0.026407914701849222, "epoch": 0.03723340161465237, "frac_reward_zero_std": 0.0, "grad_norm": 0.8467232584953308, "learning_rate": 7.193939393939394e-06, "loss": -0.014, "num_tokens": 2075400.0, "reward": 0.4593149423599243, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7232142686843872, "reward_count_adherence_std": 0.02525380253791809, "reward_meter_mean": 0.9979950189590454, "reward_meter_std": 0.0002653496921993792, "reward_repeat_penalty_mean": 0.6365914344787598, "reward_repeat_penalty_std": 0.01973436214029789, "reward_std": 0.01691724918782711, "reward_total_composite_mean": 0.4593149423599243, "reward_total_composite_std": 0.01691725291311741, "reward_total_mean": 0.4593149423599243, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7232142686843872, "rewards/count_adherence/std": 0.02525380253791809, "rewards/meter/mean": 0.9979950189590454, "rewards/meter/std": 0.0002653496921993792, "rewards/repeat_penalty/mean": 0.6365914344787598, "rewards/repeat_penalty/std": 0.01973436214029789, "rewards/total_composite/mean": 0.4593149423599243, "rewards/total_composite/std": 0.01691725291311741, "sampling/importance_sampling_ratio/max": 1.8701292276382446, "sampling/importance_sampling_ratio/mean": 1.0001847743988037, "sampling/importance_sampling_ratio/min": 0.17939433455467224, "sampling/sampling_logp_difference/max": 1.7181689739227295, "sampling/sampling_logp_difference/mean": 0.0067790052853524685, "step": 927 }, { "clip_ratio/high_max": 0.008021060610190034, "clip_ratio/high_mean": 0.008021060610190034, "clip_ratio/low_mean": 0.0015723269898444414, "clip_ratio/low_min": 0.0015723269898444414, "clip_ratio/region_mean": 0.009593387600034475, "completions/clipped_ratio": 0.0, "completions/max_length": 161.0, "completions/max_terminated_length": 161.0, "completions/mean_length": 154.375, "completions/mean_terminated_length": 154.375, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "entropy": 0.04184746090322733, "epoch": 0.037273567096437324, "frac_reward_zero_std": 0.0, "grad_norm": 1.9796561002731323, "learning_rate": 7.1909090909090914e-06, "loss": 0.0067, "num_tokens": 2078027.0, "reward": 0.645013689994812, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8792700171470642, "reward_meter_std": 0.2660437226295471, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.20092782378196716, "reward_total_composite_mean": 0.645013689994812, "reward_total_composite_std": 0.20092783868312836, "reward_total_mean": 0.645013689994812, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8792700171470642, "rewards/meter/std": 0.2660437226295471, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.645013689994812, "rewards/total_composite/std": 0.20092783868312836, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9983150959014893, "sampling/importance_sampling_ratio/min": 0.19323588907718658, "sampling/sampling_logp_difference/max": 1.643843650817871, "sampling/sampling_logp_difference/mean": 0.013188260607421398, "step": 928 }, { "clip_ratio/high_max": 0.002923976629972458, "clip_ratio/high_mean": 0.002923976629972458, "clip_ratio/low_mean": 0.003659270762000233, "clip_ratio/low_min": 0.003659270762000233, "clip_ratio/region_mean": 0.006583247391972691, "completions/clipped_ratio": 0.0, "completions/max_length": 171.0, "completions/max_terminated_length": 171.0, "completions/mean_length": 170.875, "completions/mean_terminated_length": 170.875, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.03398467996157706, "epoch": 0.03731373257822228, "frac_reward_zero_std": 0.0, "grad_norm": 2.6719958782196045, "learning_rate": 7.187878787878788e-06, "loss": 0.0038, "num_tokens": 2080954.0, "reward": 0.8040474057197571, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981188178062439, "reward_meter_std": 0.0002052018535323441, "reward_repeat_penalty_mean": 0.8055555820465088, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05147679150104523, "reward_total_composite_mean": 0.8040474057197571, "reward_total_composite_std": 0.051476798951625824, "reward_total_mean": 0.8040474057197571, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981188178062439, "rewards/meter/std": 0.0002052018535323441, "rewards/repeat_penalty/mean": 0.8055555820465088, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.8040474057197571, "rewards/total_composite/std": 0.051476798951625824, "sampling/importance_sampling_ratio/max": 1.3845103979110718, "sampling/importance_sampling_ratio/mean": 0.9989221692085266, "sampling/importance_sampling_ratio/min": 0.17145249247550964, "sampling/sampling_logp_difference/max": 1.7634490728378296, "sampling/sampling_logp_difference/mean": 0.008916568011045456, "step": 929 }, { "clip_ratio/high_max": 0.016671410761773586, "clip_ratio/high_mean": 0.016671410761773586, "clip_ratio/low_mean": 0.016923486720770597, "clip_ratio/low_min": 0.016923486720770597, "clip_ratio/region_mean": 0.033594897482544184, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 77.5, "completions/mean_terminated_length": 77.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.26597489789128304, "epoch": 0.03735389806000723, "frac_reward_zero_std": 0.0, "grad_norm": 5.791079998016357, "learning_rate": 7.184848484848486e-06, "loss": 0.0414, "num_tokens": 2082718.0, "reward": 0.9255967140197754, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9255967140197754, "reward_meter_std": 0.06689538806676865, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06689538061618805, "reward_total_composite_mean": 0.9255967140197754, "reward_total_composite_std": 0.06689538806676865, "reward_total_mean": 0.9255967140197754, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9255967140197754, "rewards/meter/std": 0.06689538806676865, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9255967140197754, "rewards/total_composite/std": 0.06689538806676865, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0131640434265137, "sampling/importance_sampling_ratio/min": 0.30976709723472595, "sampling/sampling_logp_difference/max": 1.1719346046447754, "sampling/sampling_logp_difference/mean": 0.04519684612751007, "step": 930 }, { "clip_ratio/high_max": 0.009181037603411824, "clip_ratio/high_mean": 0.009181037603411824, "clip_ratio/low_mean": 0.00041946308920159936, "clip_ratio/low_min": 0.00041946308920159936, "clip_ratio/region_mean": 0.009600500692613423, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 309.625, "completions/mean_terminated_length": 309.625, "completions/min_length": 289.0, "completions/min_terminated_length": 289.0, "entropy": 0.031651133904233575, "epoch": 0.037394063541792186, "frac_reward_zero_std": 0.0, "grad_norm": 0.8847100734710693, "learning_rate": 7.181818181818182e-06, "loss": -0.0247, "num_tokens": 2086955.0, "reward": 0.4331192076206207, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7983001470565796, "reward_meter_std": 0.3538459241390228, "reward_repeat_penalty_mean": 0.6083333492279053, "reward_repeat_penalty_std": 0.0235702246427536, "reward_std": 0.19441883265972137, "reward_total_composite_mean": 0.4331192076206207, "reward_total_composite_std": 0.19441884756088257, "reward_total_mean": 0.4331192076206207, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7983001470565796, "rewards/meter/std": 0.3538459241390228, "rewards/repeat_penalty/mean": 0.6083333492279053, "rewards/repeat_penalty/std": 0.0235702246427536, "rewards/total_composite/mean": 0.4331192076206207, "rewards/total_composite/std": 0.19441884756088257, "sampling/importance_sampling_ratio/max": 1.7735785245895386, "sampling/importance_sampling_ratio/mean": 0.9997681379318237, "sampling/importance_sampling_ratio/min": 0.3396088182926178, "sampling/sampling_logp_difference/max": 1.079960823059082, "sampling/sampling_logp_difference/mean": 0.008012884296476841, "step": 931 }, { "clip_ratio/high_max": 0.010114155360497534, "clip_ratio/high_mean": 0.010114155360497534, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/region_mean": 0.011826484114862978, "completions/clipped_ratio": 0.0, "completions/max_length": 150.0, "completions/max_terminated_length": 150.0, "completions/mean_length": 148.5, "completions/mean_terminated_length": 148.5, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.037923737429082394, "epoch": 0.03743422902357714, "frac_reward_zero_std": 0.0, "grad_norm": 2.1663684844970703, "learning_rate": 7.17878787878788e-06, "loss": -0.0003, "num_tokens": 2089791.0, "reward": 0.5223252773284912, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7308982610702515, "reward_meter_std": 0.4457210898399353, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.317903995513916, "reward_total_composite_mean": 0.5223252773284912, "reward_total_composite_std": 0.317903995513916, "reward_total_mean": 0.5223252773284912, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7308982610702515, "rewards/meter/std": 0.4457210898399353, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.5223252773284912, "rewards/total_composite/std": 0.317903995513916, "sampling/importance_sampling_ratio/max": 1.8277068138122559, "sampling/importance_sampling_ratio/mean": 0.9969629049301147, "sampling/importance_sampling_ratio/min": 0.3468512296676636, "sampling/sampling_logp_difference/max": 1.0588593482971191, "sampling/sampling_logp_difference/mean": 0.012427828274667263, "step": 932 }, { "clip_ratio/high_max": 0.009302935097366571, "clip_ratio/high_mean": 0.009302935097366571, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.009302935097366571, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 53.25, "completions/mean_terminated_length": 53.25, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.1050838534720242, "epoch": 0.037474394505362094, "frac_reward_zero_std": 0.0, "grad_norm": 8.870975494384766, "learning_rate": 7.175757575757576e-06, "loss": -0.0034, "num_tokens": 2091401.0, "reward": 0.9159232378005981, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9159232378005981, "reward_meter_std": 0.05563880503177643, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05563879758119583, "reward_total_composite_mean": 0.9159232378005981, "reward_total_composite_std": 0.05563880503177643, "reward_total_mean": 0.9159232378005981, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9159232378005981, "rewards/meter/std": 0.05563880503177643, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9159232378005981, "rewards/total_composite/std": 0.05563880503177643, "sampling/importance_sampling_ratio/max": 1.4143028259277344, "sampling/importance_sampling_ratio/mean": 1.0049976110458374, "sampling/importance_sampling_ratio/min": 0.26748713850975037, "sampling/sampling_logp_difference/max": 1.3186838626861572, "sampling/sampling_logp_difference/mean": 0.023871900513768196, "step": 933 }, { "clip_ratio/high_max": 0.016192146576941013, "clip_ratio/high_mean": 0.016192146576941013, "clip_ratio/low_mean": 0.01640054234303534, "clip_ratio/low_min": 0.01640054234303534, "clip_ratio/region_mean": 0.032592688919976354, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.1644162703305483, "epoch": 0.03751455998714705, "frac_reward_zero_std": 0.0, "grad_norm": 6.214112281799316, "learning_rate": 7.172727272727273e-06, "loss": 0.0017, "num_tokens": 2093250.0, "reward": 0.40722712874412537, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.40722712874412537, "reward_meter_std": 0.28478607535362244, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.28478607535362244, "reward_total_composite_mean": 0.40722712874412537, "reward_total_composite_std": 0.28478607535362244, "reward_total_mean": 0.40722712874412537, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.40722712874412537, "rewards/meter/std": 0.28478607535362244, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.40722712874412537, "rewards/total_composite/std": 0.28478607535362244, "sampling/importance_sampling_ratio/max": 1.4551042318344116, "sampling/importance_sampling_ratio/mean": 0.9984850883483887, "sampling/importance_sampling_ratio/min": 0.1503974199295044, "sampling/sampling_logp_difference/max": 1.8944740295410156, "sampling/sampling_logp_difference/mean": 0.03458065912127495, "step": 934 }, { "clip_ratio/high_max": 0.00390625, "clip_ratio/high_mean": 0.00390625, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/region_mean": 0.015269886702299118, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.625, "completions/mean_terminated_length": 32.625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.051071187015622854, "epoch": 0.037554725468932, "frac_reward_zero_std": 0.0, "grad_norm": 7.902743816375732, "learning_rate": 7.16969696969697e-06, "loss": 0.0203, "num_tokens": 2094727.0, "reward": 0.9710537195205688, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9710537195205688, "reward_meter_std": 0.003800545586273074, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003800547681748867, "reward_total_composite_mean": 0.9710537195205688, "reward_total_composite_std": 0.003800545586273074, "reward_total_mean": 0.9710537195205688, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9710537195205688, "rewards/meter/std": 0.003800545586273074, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9710537195205688, "rewards/total_composite/std": 0.003800545586273074, "sampling/importance_sampling_ratio/max": 1.517050862312317, "sampling/importance_sampling_ratio/mean": 1.0037425756454468, "sampling/importance_sampling_ratio/min": 0.7209804058074951, "sampling/sampling_logp_difference/max": 0.41676831245422363, "sampling/sampling_logp_difference/mean": 0.009311332367360592, "step": 935 }, { "clip_ratio/high_max": 0.0048290317645296454, "clip_ratio/high_mean": 0.0048290317645296454, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/region_mean": 0.00639153178781271, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 77.5, "completions/mean_terminated_length": 77.5, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.09636378474533558, "epoch": 0.037594890950716955, "frac_reward_zero_std": 0.0, "grad_norm": 12.27761173248291, "learning_rate": 7.166666666666667e-06, "loss": 0.008, "num_tokens": 2096787.0, "reward": 0.9931102991104126, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9931102991104126, "reward_meter_std": 0.006781714037060738, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006781699601560831, "reward_total_composite_mean": 0.9931102991104126, "reward_total_composite_std": 0.006781714037060738, "reward_total_mean": 0.9931102991104126, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9931102991104126, "rewards/meter/std": 0.006781714037060738, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931102991104126, "rewards/total_composite/std": 0.006781714037060738, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001671314239502, "sampling/importance_sampling_ratio/min": 0.4686746895313263, "sampling/sampling_logp_difference/max": 0.852900505065918, "sampling/sampling_logp_difference/mean": 0.017138954252004623, "step": 936 }, { "clip_ratio/high_max": 0.014654986094683409, "clip_ratio/high_mean": 0.014654986094683409, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/region_mean": 0.018278174567967653, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.75, "completions/mean_terminated_length": 68.75, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.0943189668469131, "epoch": 0.03763505643250191, "frac_reward_zero_std": 0.0, "grad_norm": 9.313652992248535, "learning_rate": 7.163636363636363e-06, "loss": 0.0106, "num_tokens": 2098529.0, "reward": 0.9961028099060059, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961028099060059, "reward_meter_std": 0.0017646643100306392, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001764659769833088, "reward_total_composite_mean": 0.9961028099060059, "reward_total_composite_std": 0.0017646643100306392, "reward_total_mean": 0.9961028099060059, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961028099060059, "rewards/meter/std": 0.0017646643100306392, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961028099060059, "rewards/total_composite/std": 0.0017646643100306392, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0019060373306274, "sampling/importance_sampling_ratio/min": 0.25915029644966125, "sampling/sampling_logp_difference/max": 1.3503470420837402, "sampling/sampling_logp_difference/mean": 0.028187252581119537, "step": 937 }, { "clip_ratio/high_max": 0.004902798566035926, "clip_ratio/high_mean": 0.004902798566035926, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.004902798566035926, "completions/clipped_ratio": 0.0, "completions/max_length": 155.0, "completions/max_terminated_length": 155.0, "completions/mean_length": 150.25, "completions/mean_terminated_length": 150.25, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.02918590500485152, "epoch": 0.03767522191428686, "frac_reward_zero_std": 0.0, "grad_norm": 2.60717511177063, "learning_rate": 7.1606060606060615e-06, "loss": -0.0077, "num_tokens": 2101075.0, "reward": 0.686731219291687, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9614237546920776, "reward_meter_std": 0.02061135321855545, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014722414314746857, "reward_total_composite_mean": 0.686731219291687, "reward_total_composite_std": 0.014722409658133984, "reward_total_mean": 0.686731219291687, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9614237546920776, "rewards/meter/std": 0.02061135321855545, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.686731219291687, "rewards/total_composite/std": 0.014722409658133984, "sampling/importance_sampling_ratio/max": 1.5528877973556519, "sampling/importance_sampling_ratio/mean": 1.0006721019744873, "sampling/importance_sampling_ratio/min": 0.2901509404182434, "sampling/sampling_logp_difference/max": 1.237354040145874, "sampling/sampling_logp_difference/mean": 0.005851758643984795, "step": 938 }, { "clip_ratio/high_max": 0.013518452877178788, "clip_ratio/high_mean": 0.013518452877178788, "clip_ratio/low_mean": 0.017079579178243876, "clip_ratio/low_min": 0.017079579178243876, "clip_ratio/region_mean": 0.030598032055422664, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 36.5, "completions/mean_terminated_length": 36.5, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.13651060685515404, "epoch": 0.03771538739607182, "frac_reward_zero_std": 0.0, "grad_norm": 12.418314933776855, "learning_rate": 7.157575757575758e-06, "loss": -0.0037, "num_tokens": 2102575.0, "reward": 0.21266832947731018, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.21266832947731018, "reward_meter_std": 0.20291058719158173, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.20291060209274292, "reward_total_composite_mean": 0.21266832947731018, "reward_total_composite_std": 0.20291058719158173, "reward_total_mean": 0.21266832947731018, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.21266832947731018, "rewards/meter/std": 0.20291058719158173, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.21266832947731018, "rewards/total_composite/std": 0.20291058719158173, "sampling/importance_sampling_ratio/max": 1.8581043481826782, "sampling/importance_sampling_ratio/mean": 1.0057456493377686, "sampling/importance_sampling_ratio/min": 0.2693828046321869, "sampling/sampling_logp_difference/max": 1.311621904373169, "sampling/sampling_logp_difference/mean": 0.03699643164873123, "step": 939 }, { "clip_ratio/high_max": 0.016339467372745275, "clip_ratio/high_mean": 0.016339467372745275, "clip_ratio/low_mean": 0.0020325202494859695, "clip_ratio/low_min": 0.0020325202494859695, "clip_ratio/region_mean": 0.018371987622231245, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 115.875, "completions/mean_terminated_length": 115.875, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.044212615583091974, "epoch": 0.03775555287785677, "frac_reward_zero_std": 0.0, "grad_norm": 4.8500847816467285, "learning_rate": 7.154545454545455e-06, "loss": 0.023, "num_tokens": 2104942.0, "reward": 0.7747991681098938, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.968498945236206, "reward_meter_std": 0.029964042827486992, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.023971248418092728, "reward_total_composite_mean": 0.7747991681098938, "reward_total_composite_std": 0.02397123910486698, "reward_total_mean": 0.7747991681098938, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.968498945236206, "rewards/meter/std": 0.029964042827486992, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7747991681098938, "rewards/total_composite/std": 0.02397123910486698, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9974149465560913, "sampling/importance_sampling_ratio/min": 0.3408257067203522, "sampling/sampling_logp_difference/max": 1.0763840675354004, "sampling/sampling_logp_difference/mean": 0.0124904103577137, "step": 940 }, { "clip_ratio/high_max": 0.008275165455415845, "clip_ratio/high_mean": 0.008275165455415845, "clip_ratio/low_mean": 0.011406926438212395, "clip_ratio/low_min": 0.011406926438212395, "clip_ratio/region_mean": 0.01968209189362824, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 75.75, "completions/mean_terminated_length": 75.75, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.08911147527396679, "epoch": 0.037795718359641725, "frac_reward_zero_std": 0.0, "grad_norm": 3.194761276245117, "learning_rate": 7.151515151515152e-06, "loss": 0.0096, "num_tokens": 2106780.0, "reward": 0.7389180660247803, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7696533203125, "reward_meter_std": 0.264241486787796, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.2821867763996124, "reward_total_composite_mean": 0.7389180660247803, "reward_total_composite_std": 0.2821867763996124, "reward_total_mean": 0.7389180660247803, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7696533203125, "rewards/meter/std": 0.264241486787796, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.7389180660247803, "rewards/total_composite/std": 0.2821867763996124, "sampling/importance_sampling_ratio/max": 1.8680615425109863, "sampling/importance_sampling_ratio/mean": 1.0013823509216309, "sampling/importance_sampling_ratio/min": 0.3468180000782013, "sampling/sampling_logp_difference/max": 1.0589550733566284, "sampling/sampling_logp_difference/mean": 0.02568567357957363, "step": 941 }, { "clip_ratio/high_max": 0.016215793788433075, "clip_ratio/high_mean": 0.016215793788433075, "clip_ratio/low_mean": 0.006667852168902755, "clip_ratio/low_min": 0.006667852168902755, "clip_ratio/region_mean": 0.02288364595733583, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 38.625, "completions/mean_terminated_length": 38.625, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.15884944330900908, "epoch": 0.03783588384142668, "frac_reward_zero_std": 0.0, "grad_norm": 11.929455757141113, "learning_rate": 7.148484848484849e-06, "loss": -0.0086, "num_tokens": 2108257.0, "reward": 0.7800159454345703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7800159454345703, "reward_meter_std": 0.17752352356910706, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.17752352356910706, "reward_total_composite_mean": 0.7800159454345703, "reward_total_composite_std": 0.17752352356910706, "reward_total_mean": 0.7800159454345703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7800159454345703, "rewards/meter/std": 0.17752352356910706, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7800159454345703, "rewards/total_composite/std": 0.17752352356910706, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0109392404556274, "sampling/importance_sampling_ratio/min": 0.5866755843162537, "sampling/sampling_logp_difference/max": 1.4989476203918457, "sampling/sampling_logp_difference/mean": 0.03380492329597473, "step": 942 }, { "clip_ratio/high_max": 0.015178571688011289, "clip_ratio/high_mean": 0.015178571688011289, "clip_ratio/low_mean": 0.006253908621147275, "clip_ratio/low_min": 0.006253908621147275, "clip_ratio/region_mean": 0.021432480309158564, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 41.125, "completions/mean_terminated_length": 41.125, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.09372959332540631, "epoch": 0.03787604932321163, "frac_reward_zero_std": 0.0, "grad_norm": 6.928620338439941, "learning_rate": 7.145454545454547e-06, "loss": -0.0148, "num_tokens": 2109890.0, "reward": 0.9855058193206787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9855058193206787, "reward_meter_std": 0.010752059519290924, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010752064175903797, "reward_total_composite_mean": 0.9855058193206787, "reward_total_composite_std": 0.010752059519290924, "reward_total_mean": 0.9855058193206787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9855058193206787, "rewards/meter/std": 0.010752059519290924, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9855058193206787, "rewards/total_composite/std": 0.010752059519290924, "sampling/importance_sampling_ratio/max": 1.2362208366394043, "sampling/importance_sampling_ratio/mean": 0.9965353012084961, "sampling/importance_sampling_ratio/min": 0.326887309551239, "sampling/sampling_logp_difference/max": 1.1181397438049316, "sampling/sampling_logp_difference/mean": 0.01714111864566803, "step": 943 }, { "clip_ratio/high_max": 0.01517849147785455, "clip_ratio/high_mean": 0.01517849147785455, "clip_ratio/low_mean": 0.004050420364364982, "clip_ratio/low_min": 0.004050420364364982, "clip_ratio/region_mean": 0.01922891184221953, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 123.0, "completions/mean_terminated_length": 123.0, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.06403588084504008, "epoch": 0.03791621480499659, "frac_reward_zero_std": 0.0, "grad_norm": 4.29167366027832, "learning_rate": 7.142424242424243e-06, "loss": 0.0057, "num_tokens": 2112370.0, "reward": 0.5850111842155457, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8062102794647217, "reward_meter_std": 0.18132120370864868, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.11425027996301651, "reward_total_composite_mean": 0.5850111842155457, "reward_total_composite_std": 0.11425027996301651, "reward_total_mean": 0.5850111842155457, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8062102794647217, "rewards/meter/std": 0.18132120370864868, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.5850111842155457, "rewards/total_composite/std": 0.11425027996301651, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.994314432144165, "sampling/importance_sampling_ratio/min": 0.1793576329946518, "sampling/sampling_logp_difference/max": 1.7183735370635986, "sampling/sampling_logp_difference/mean": 0.026136090978980064, "step": 944 }, { "clip_ratio/high_max": 0.011334090260788798, "clip_ratio/high_mean": 0.011334090260788798, "clip_ratio/low_mean": 0.002040850231423974, "clip_ratio/low_min": 0.002040850231423974, "clip_ratio/region_mean": 0.013374940492212772, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 120.625, "completions/mean_terminated_length": 120.625, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.05396859138272703, "epoch": 0.03795638028678154, "frac_reward_zero_std": 0.0, "grad_norm": 4.3193440437316895, "learning_rate": 7.1393939393939405e-06, "loss": 0.0177, "num_tokens": 2114655.0, "reward": 0.6976097822189331, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9766536355018616, "reward_meter_std": 0.021832915022969246, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.0, "reward_std": 0.015594931319355965, "reward_total_composite_mean": 0.6976097822189331, "reward_total_composite_std": 0.015594935044646263, "reward_total_mean": 0.6976097822189331, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9766536355018616, "rewards/meter/std": 0.021832915022969246, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6976097822189331, "rewards/total_composite/std": 0.015594935044646263, "sampling/importance_sampling_ratio/max": 1.6762388944625854, "sampling/importance_sampling_ratio/mean": 0.9987999200820923, "sampling/importance_sampling_ratio/min": 0.2696869373321533, "sampling/sampling_logp_difference/max": 1.3104934692382812, "sampling/sampling_logp_difference/mean": 0.015354371629655361, "step": 945 }, { "clip_ratio/high_max": 0.008308789343573153, "clip_ratio/high_mean": 0.008308789343573153, "clip_ratio/low_mean": 0.00222885946277529, "clip_ratio/low_min": 0.00222885946277529, "clip_ratio/region_mean": 0.010537648806348443, "completions/clipped_ratio": 0.0, "completions/max_length": 171.0, "completions/max_terminated_length": 171.0, "completions/mean_length": 166.0, "completions/mean_terminated_length": 166.0, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.03645319910719991, "epoch": 0.037996545768566495, "frac_reward_zero_std": 0.0, "grad_norm": 1.7848657369613647, "learning_rate": 7.136363636363637e-06, "loss": 0.0042, "num_tokens": 2117343.0, "reward": 0.6401993036270142, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958733320236206, "reward_meter_std": 0.00040963603532873094, "reward_repeat_penalty_mean": 0.6428571939468384, "reward_repeat_penalty_std": 0.1322600245475769, "reward_std": 0.13169293105602264, "reward_total_composite_mean": 0.6401993036270142, "reward_total_composite_std": 0.13169293105602264, "reward_total_mean": 0.6401993036270142, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958733320236206, "rewards/meter/std": 0.00040963603532873094, "rewards/repeat_penalty/mean": 0.6428571939468384, "rewards/repeat_penalty/std": 0.1322600245475769, "rewards/total_composite/mean": 0.6401993036270142, "rewards/total_composite/std": 0.13169293105602264, "sampling/importance_sampling_ratio/max": 1.6956576108932495, "sampling/importance_sampling_ratio/mean": 0.9997925162315369, "sampling/importance_sampling_ratio/min": 0.35387131571769714, "sampling/sampling_logp_difference/max": 1.0388219356536865, "sampling/sampling_logp_difference/mean": 0.009576267562806606, "step": 946 }, { "clip_ratio/high_max": 0.0017141569405794144, "clip_ratio/high_mean": 0.0017141569405794144, "clip_ratio/low_mean": 0.0021205995872151107, "clip_ratio/low_min": 0.0021205995872151107, "clip_ratio/region_mean": 0.003834756527794525, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 293.5, "completions/mean_terminated_length": 293.5, "completions/min_length": 287.0, "completions/min_terminated_length": 287.0, "entropy": 0.021893693367019296, "epoch": 0.03803671125035145, "frac_reward_zero_std": 0.0, "grad_norm": 1.0996932983398438, "learning_rate": 7.133333333333334e-06, "loss": -0.004, "num_tokens": 2121363.0, "reward": 0.37909868359565735, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_meter_mean": 0.9976366758346558, "reward_meter_std": 0.0012180638732388616, "reward_repeat_penalty_mean": 0.39835166931152344, "reward_repeat_penalty_std": 0.22086545825004578, "reward_std": 0.22086749970912933, "reward_total_composite_mean": 0.37909868359565735, "reward_total_composite_std": 0.22086751461029053, "reward_total_mean": 0.37909868359565735, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/meter/mean": 0.9976366758346558, "rewards/meter/std": 0.0012180638732388616, "rewards/repeat_penalty/mean": 0.39835166931152344, "rewards/repeat_penalty/std": 0.22086545825004578, "rewards/total_composite/mean": 0.37909868359565735, "rewards/total_composite/std": 0.22086751461029053, "sampling/importance_sampling_ratio/max": 1.6547387838363647, "sampling/importance_sampling_ratio/mean": 1.0007612705230713, "sampling/importance_sampling_ratio/min": 0.4687469005584717, "sampling/sampling_logp_difference/max": 0.7576923370361328, "sampling/sampling_logp_difference/mean": 0.0038092564791440964, "step": 947 }, { "clip_ratio/high_max": 0.008802816737443209, "clip_ratio/high_mean": 0.008802816737443209, "clip_ratio/low_mean": 0.0072228144854307175, "clip_ratio/low_min": 0.0072228144854307175, "clip_ratio/region_mean": 0.016025631222873926, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 70.25, "completions/mean_terminated_length": 70.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.1040426753461361, "epoch": 0.0380768767321364, "frac_reward_zero_std": 0.0, "grad_norm": 6.672044277191162, "learning_rate": 7.130303030303031e-06, "loss": -0.012, "num_tokens": 2123173.0, "reward": 0.9970372915267944, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970372915267944, "reward_meter_std": 0.0017233057878911495, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001723314169794321, "reward_total_composite_mean": 0.9970372915267944, "reward_total_composite_std": 0.0017233057878911495, "reward_total_mean": 0.9970372915267944, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970372915267944, "rewards/meter/std": 0.0017233057878911495, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970372915267944, "rewards/total_composite/std": 0.0017233057878911495, "sampling/importance_sampling_ratio/max": 1.8399385213851929, "sampling/importance_sampling_ratio/mean": 1.0034829378128052, "sampling/importance_sampling_ratio/min": 0.43877550959587097, "sampling/sampling_logp_difference/max": 0.823767364025116, "sampling/sampling_logp_difference/mean": 0.02216244488954544, "step": 948 }, { "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/low_mean": 0.006048386916518211, "clip_ratio/low_min": 0.006048386916518211, "clip_ratio/region_mean": 0.010146747343242168, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 60.625, "completions/mean_terminated_length": 60.625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.03217482636682689, "epoch": 0.03811704221392136, "frac_reward_zero_std": 0.0, "grad_norm": 1.9708746671676636, "learning_rate": 7.127272727272728e-06, "loss": 0.0108, "num_tokens": 2125026.0, "reward": 0.5285885334014893, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7747896909713745, "reward_meter_std": 0.38887178897857666, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.23691309988498688, "reward_total_composite_mean": 0.5285885334014893, "reward_total_composite_std": 0.23691311478614807, "reward_total_mean": 0.5285885334014893, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7747896909713745, "rewards/meter/std": 0.38887178897857666, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.5285885334014893, "rewards/total_composite/std": 0.23691311478614807, "sampling/importance_sampling_ratio/max": 1.2806702852249146, "sampling/importance_sampling_ratio/mean": 1.0037261247634888, "sampling/importance_sampling_ratio/min": 0.48986631631851196, "sampling/sampling_logp_difference/max": 0.7136227488517761, "sampling/sampling_logp_difference/mean": 0.0069820573553442955, "step": 949 }, { "clip_ratio/high_max": 0.007258509169332683, "clip_ratio/high_mean": 0.007258509169332683, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007258509169332683, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 120.125, "completions/mean_terminated_length": 120.125, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.030512073542922735, "epoch": 0.03815720769570631, "frac_reward_zero_std": 0.0, "grad_norm": 3.3005993366241455, "learning_rate": 7.124242424242424e-06, "loss": 0.0008, "num_tokens": 2127299.0, "reward": 0.7464566826820374, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9637161493301392, "reward_meter_std": 0.08138205856084824, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.09103825688362122, "reward_total_composite_mean": 0.7464566826820374, "reward_total_composite_std": 0.09103824943304062, "reward_total_mean": 0.7464566826820374, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9637161493301392, "rewards/meter/std": 0.08138205856084824, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7464566826820374, "rewards/total_composite/std": 0.09103824943304062, "sampling/importance_sampling_ratio/max": 1.577369213104248, "sampling/importance_sampling_ratio/mean": 0.996985137462616, "sampling/importance_sampling_ratio/min": 0.2125377207994461, "sampling/sampling_logp_difference/max": 1.548635721206665, "sampling/sampling_logp_difference/mean": 0.010762141086161137, "step": 950 }, { "epoch": 0.03815720769570631, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 376.0769230769231, "eval_completions/max_terminated_length": 376.0769230769231, "eval_completions/mean_length": 205.85576923076923, "eval_completions/mean_terminated_length": 205.85576923076923, "eval_completions/min_length": 61.53846153846154, "eval_completions/min_terminated_length": 61.53846153846154, "eval_entropy": 0.04716356213276203, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 2127299.0, "eval_reward": 0.41787450359417844, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.9363291034331689, "eval_reward_count_adherence_std": 0.09608005073208076, "eval_reward_meter_mean": 0.6758117584081796, "eval_reward_meter_std": 0.3830513243491833, "eval_reward_repeat_penalty_mean": 0.6863612211667575, "eval_reward_repeat_penalty_std": 0.2006635952454347, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.41787450359417844, "eval_reward_total_composite_std": 0.3079184889793396, "eval_reward_total_mean": 0.41787450359417844, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.9363291034331689, "eval_rewards/count_adherence/std": 0.09608005073208076, "eval_rewards/meter/mean": 0.6758117584081796, "eval_rewards/meter/std": 0.3830513243491833, "eval_rewards/repeat_penalty/mean": 0.6863612211667575, "eval_rewards/repeat_penalty/std": 0.2006635952454347, "eval_rewards/total_composite/mean": 0.41787450359417844, "eval_rewards/total_composite/std": 0.3079184889793396, "eval_runtime": 71.2662, "eval_samples_per_second": 1.459, "eval_sampling/importance_sampling_ratio/max": 1.2753138267076933, "eval_sampling/importance_sampling_ratio/mean": 1.0011087380922759, "eval_sampling/importance_sampling_ratio/min": 0.5113821442310627, "eval_sampling/sampling_logp_difference/max": 0.7134297077472394, "eval_sampling/sampling_logp_difference/mean": 0.0055299800498267776, "eval_steps_per_second": 0.182, "step": 950 }, { "clip_ratio/high_max": 0.011278833262622356, "clip_ratio/high_mean": 0.011278833262622356, "clip_ratio/low_mean": 0.00940205657389015, "clip_ratio/low_min": 0.00940205657389015, "clip_ratio/region_mean": 0.020680889836512506, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.21755263209342957, "epoch": 0.038197373177491264, "frac_reward_zero_std": 0.0, "grad_norm": 4.512762069702148, "learning_rate": 7.121212121212122e-06, "loss": 0.0095, "num_tokens": 2129201.0, "reward": 0.2129497528076172, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2129497528076172, "reward_meter_std": 0.24503695964813232, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.24503694474697113, "reward_total_composite_mean": 0.2129497528076172, "reward_total_composite_std": 0.24503695964813232, "reward_total_mean": 0.2129497528076172, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2129497528076172, "rewards/meter/std": 0.24503695964813232, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2129497528076172, "rewards/total_composite/std": 0.24503695964813232, "sampling/importance_sampling_ratio/max": 1.650530457496643, "sampling/importance_sampling_ratio/mean": 1.001453161239624, "sampling/importance_sampling_ratio/min": 0.2619650065898895, "sampling/sampling_logp_difference/max": 1.3395442962646484, "sampling/sampling_logp_difference/mean": 0.02744590863585472, "step": 951 }, { "clip_ratio/high_max": 0.0016673028003424406, "clip_ratio/high_mean": 0.0016673028003424406, "clip_ratio/low_mean": 0.0016632305341772735, "clip_ratio/low_min": 0.0016632305341772735, "clip_ratio/region_mean": 0.003330533334519714, "completions/clipped_ratio": 0.0, "completions/max_length": 470.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 451.0, "completions/mean_terminated_length": 451.0, "completions/min_length": 446.0, "completions/min_terminated_length": 446.0, "entropy": 0.013566843350417912, "epoch": 0.03823753865927622, "frac_reward_zero_std": 0.0, "grad_norm": 0.9411554336547852, "learning_rate": 7.118181818181819e-06, "loss": -0.0019, "num_tokens": 2134537.0, "reward": 0.17538189888000488, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6315789222717285, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.49129414558410645, "reward_meter_std": 0.5206316113471985, "reward_repeat_penalty_mean": 0.5652173757553101, "reward_repeat_penalty_std": 0.0, "reward_std": 0.18585477769374847, "reward_total_composite_mean": 0.17538189888000488, "reward_total_composite_std": 0.18585476279258728, "reward_total_mean": 0.17538189888000488, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6315789222717285, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.49129414558410645, "rewards/meter/std": 0.5206316113471985, "rewards/repeat_penalty/mean": 0.5652173757553101, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.17538189888000488, "rewards/total_composite/std": 0.18585476279258728, "sampling/importance_sampling_ratio/max": 1.997017741203308, "sampling/importance_sampling_ratio/mean": 1.0005483627319336, "sampling/importance_sampling_ratio/min": 0.04932800680398941, "sampling/sampling_logp_difference/max": 3.009263277053833, "sampling/sampling_logp_difference/mean": 0.004929275717586279, "step": 952 }, { "clip_ratio/high_max": 0.00769954826682806, "clip_ratio/high_mean": 0.00769954826682806, "clip_ratio/low_mean": 0.02908552496228367, "clip_ratio/low_min": 0.02908552496228367, "clip_ratio/region_mean": 0.03678507322911173, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 80.5, "completions/mean_terminated_length": 80.5, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.09662660211324692, "epoch": 0.03827770414106117, "frac_reward_zero_std": 0.0, "grad_norm": 7.106107711791992, "learning_rate": 7.115151515151516e-06, "loss": 0.0166, "num_tokens": 2136613.0, "reward": 0.8412514925003052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9900895357131958, "reward_meter_std": 0.006374072283506393, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.08799233287572861, "reward_total_composite_mean": 0.8412514925003052, "reward_total_composite_std": 0.08799233287572861, "reward_total_mean": 0.8412514925003052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9900895357131958, "rewards/meter/std": 0.006374072283506393, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8412514925003052, "rewards/total_composite/std": 0.08799233287572861, "sampling/importance_sampling_ratio/max": 1.8997235298156738, "sampling/importance_sampling_ratio/mean": 1.0007519721984863, "sampling/importance_sampling_ratio/min": 0.0587230883538723, "sampling/sampling_logp_difference/max": 2.8349223136901855, "sampling/sampling_logp_difference/mean": 0.03556607663631439, "step": 953 }, { "clip_ratio/high_max": 0.009797494392842054, "clip_ratio/high_mean": 0.009797494392842054, "clip_ratio/low_mean": 0.0235287812538445, "clip_ratio/low_min": 0.0235287812538445, "clip_ratio/region_mean": 0.033326275646686554, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 104.875, "completions/mean_terminated_length": 104.875, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.17342047207057476, "epoch": 0.038317869622846126, "frac_reward_zero_std": 0.0, "grad_norm": 3.721968650817871, "learning_rate": 7.1121212121212125e-06, "loss": 0.0104, "num_tokens": 2138876.0, "reward": 0.22144606709480286, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.28197556734085083, "reward_meter_std": 0.30284935235977173, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.24437181651592255, "reward_total_composite_mean": 0.22144606709480286, "reward_total_composite_std": 0.24437181651592255, "reward_total_mean": 0.22144606709480286, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.28197556734085083, "rewards/meter/std": 0.30284935235977173, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.22144606709480286, "rewards/total_composite/std": 0.24437181651592255, "sampling/importance_sampling_ratio/max": 1.7154247760772705, "sampling/importance_sampling_ratio/mean": 1.00920569896698, "sampling/importance_sampling_ratio/min": 0.23661062121391296, "sampling/sampling_logp_difference/max": 1.4413394927978516, "sampling/sampling_logp_difference/mean": 0.029477477073669434, "step": 954 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.005445147980935872, "clip_ratio/low_min": 0.005445147980935872, "clip_ratio/region_mean": 0.005445147980935872, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 113.75, "completions/mean_terminated_length": 113.75, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.02124447817914188, "epoch": 0.03835803510463108, "frac_reward_zero_std": 0.0, "grad_norm": 0.7482571005821228, "learning_rate": 7.10909090909091e-06, "loss": 0.009, "num_tokens": 2141066.0, "reward": 0.7753492593765259, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.969186544418335, "reward_meter_std": 0.005740889813750982, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0045927101746201515, "reward_total_composite_mean": 0.7753492593765259, "reward_total_composite_std": 0.0045927222818136215, "reward_total_mean": 0.7753492593765259, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.969186544418335, "rewards/meter/std": 0.005740889813750982, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7753492593765259, "rewards/total_composite/std": 0.0045927222818136215, "sampling/importance_sampling_ratio/max": 1.501456379890442, "sampling/importance_sampling_ratio/mean": 1.0022902488708496, "sampling/importance_sampling_ratio/min": 0.729997992515564, "sampling/sampling_logp_difference/max": 0.406435489654541, "sampling/sampling_logp_difference/mean": 0.003761922474950552, "step": 955 }, { "clip_ratio/high_max": 0.0036183277843520045, "clip_ratio/high_mean": 0.0036183277843520045, "clip_ratio/low_mean": 0.0059523810632526875, "clip_ratio/low_min": 0.0059523810632526875, "clip_ratio/region_mean": 0.009570708847604692, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 104.0, "completions/mean_terminated_length": 104.0, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.07065040059387684, "epoch": 0.038398200586416034, "frac_reward_zero_std": 0.0, "grad_norm": 3.8508245944976807, "learning_rate": 7.106060606060606e-06, "loss": 0.014, "num_tokens": 2143130.0, "reward": 0.8979662656784058, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977437853813171, "reward_meter_std": 0.0006820790586061776, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.10663473606109619, "reward_total_composite_mean": 0.8979662656784058, "reward_total_composite_std": 0.10663474351167679, "reward_total_mean": 0.8979662656784058, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977437853813171, "rewards/meter/std": 0.0006820790586061776, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8979662656784058, "rewards/total_composite/std": 0.10663474351167679, "sampling/importance_sampling_ratio/max": 1.998322606086731, "sampling/importance_sampling_ratio/mean": 1.0018119812011719, "sampling/importance_sampling_ratio/min": 0.1308285892009735, "sampling/sampling_logp_difference/max": 2.033867359161377, "sampling/sampling_logp_difference/mean": 0.026750473305583, "step": 956 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/region_mean": 0.0037313431967049837, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 60.875, "completions/mean_terminated_length": 60.875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.01487105863634497, "epoch": 0.03843836606820099, "frac_reward_zero_std": 0.0, "grad_norm": 6.492129802703857, "learning_rate": 7.103030303030304e-06, "loss": 0.033, "num_tokens": 2145169.0, "reward": 0.6475568413734436, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9347407817840576, "reward_meter_std": 0.14110930263996124, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_std": 0.02506948821246624, "reward_total_composite_mean": 0.6475568413734436, "reward_total_composite_std": 0.025069493800401688, "reward_total_mean": 0.6475568413734436, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9347407817840576, "rewards/meter/std": 0.14110930263996124, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.6475568413734436, "rewards/total_composite/std": 0.025069493800401688, "sampling/importance_sampling_ratio/max": 1.1547554731369019, "sampling/importance_sampling_ratio/mean": 0.9995921850204468, "sampling/importance_sampling_ratio/min": 0.6776925325393677, "sampling/sampling_logp_difference/max": 0.3890615701675415, "sampling/sampling_logp_difference/mean": 0.003160211257636547, "step": 957 }, { "clip_ratio/high_max": 0.010615079430863261, "clip_ratio/high_mean": 0.010615079430863261, "clip_ratio/low_mean": 0.010714286006987095, "clip_ratio/low_min": 0.010714286006987095, "clip_ratio/region_mean": 0.021329365437850356, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.10543786082416773, "epoch": 0.03847853154998594, "frac_reward_zero_std": 0.0, "grad_norm": 10.342998504638672, "learning_rate": 7.100000000000001e-06, "loss": -0.0039, "num_tokens": 2146649.0, "reward": 0.4885302782058716, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4885302782058716, "reward_meter_std": 0.35837680101394653, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35837680101394653, "reward_total_composite_mean": 0.4885302782058716, "reward_total_composite_std": 0.35837680101394653, "reward_total_mean": 0.4885302782058716, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4885302782058716, "rewards/meter/std": 0.35837680101394653, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4885302782058716, "rewards/total_composite/std": 0.35837680101394653, "sampling/importance_sampling_ratio/max": 1.507129430770874, "sampling/importance_sampling_ratio/mean": 0.9996507167816162, "sampling/importance_sampling_ratio/min": 0.25397083163261414, "sampling/sampling_logp_difference/max": 1.3705358505249023, "sampling/sampling_logp_difference/mean": 0.03277422487735748, "step": 958 }, { "clip_ratio/high_max": 0.006645367917371914, "clip_ratio/high_mean": 0.006645367917371914, "clip_ratio/low_mean": 0.001365295291179791, "clip_ratio/low_min": 0.001365295291179791, "clip_ratio/region_mean": 0.008010663208551705, "completions/clipped_ratio": 0.0, "completions/max_length": 310.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 294.5, "completions/mean_terminated_length": 294.5, "completions/min_length": 274.0, "completions/min_terminated_length": 274.0, "entropy": 0.04552340344525874, "epoch": 0.038518697031770896, "frac_reward_zero_std": 0.0, "grad_norm": 1.4002562761306763, "learning_rate": 7.096969696969698e-06, "loss": -0.0351, "num_tokens": 2150789.0, "reward": 0.4822729825973511, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977169036865234, "reward_meter_std": 0.0006972014671191573, "reward_repeat_penalty_mean": 0.4833333492279053, "reward_repeat_penalty_std": 0.216024711728096, "reward_std": 0.21558886766433716, "reward_total_composite_mean": 0.4822729825973511, "reward_total_composite_std": 0.21558886766433716, "reward_total_mean": 0.4822729825973511, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977169036865234, "rewards/meter/std": 0.0006972014671191573, "rewards/repeat_penalty/mean": 0.4833333492279053, "rewards/repeat_penalty/std": 0.216024711728096, "rewards/total_composite/mean": 0.4822729825973511, "rewards/total_composite/std": 0.21558886766433716, "sampling/importance_sampling_ratio/max": 1.689242959022522, "sampling/importance_sampling_ratio/mean": 0.9994634389877319, "sampling/importance_sampling_ratio/min": 0.3570125997066498, "sampling/sampling_logp_difference/max": 1.0299842357635498, "sampling/sampling_logp_difference/mean": 0.0083305099979043, "step": 959 }, { "clip_ratio/high_max": 0.009170350269414485, "clip_ratio/high_mean": 0.009170350269414485, "clip_ratio/low_mean": 0.009524673456326127, "clip_ratio/low_min": 0.009524673456326127, "clip_ratio/region_mean": 0.01869502372574061, "completions/clipped_ratio": 0.0, "completions/max_length": 165.0, "completions/max_terminated_length": 165.0, "completions/mean_length": 160.75, "completions/mean_terminated_length": 160.75, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.05065230838954449, "epoch": 0.03855886251355585, "frac_reward_zero_std": 0.0, "grad_norm": 6.993642807006836, "learning_rate": 7.093939393939394e-06, "loss": -0.0083, "num_tokens": 2153531.0, "reward": 0.2650865912437439, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.36679452657699585, "reward_meter_std": 0.16856136918067932, "reward_repeat_penalty_mean": 0.7222222089767456, "reward_repeat_penalty_std": 0.08399210125207901, "reward_std": 0.1350317746400833, "reward_total_composite_mean": 0.2650865912437439, "reward_total_composite_std": 0.1350317746400833, "reward_total_mean": 0.2650865912437439, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.36679452657699585, "rewards/meter/std": 0.16856136918067932, "rewards/repeat_penalty/mean": 0.7222222089767456, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.2650865912437439, "rewards/total_composite/std": 0.1350317746400833, "sampling/importance_sampling_ratio/max": 1.8480039834976196, "sampling/importance_sampling_ratio/mean": 0.9979313015937805, "sampling/importance_sampling_ratio/min": 0.00937778502702713, "sampling/sampling_logp_difference/max": 4.669411659240723, "sampling/sampling_logp_difference/mean": 0.02235065959393978, "step": 960 }, { "clip_ratio/high_max": 0.01166338287293911, "clip_ratio/high_mean": 0.01166338287293911, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.01166338287293911, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 65.125, "completions/mean_terminated_length": 65.125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.0778669067658484, "epoch": 0.038599027995340804, "frac_reward_zero_std": 0.0, "grad_norm": 5.430922508239746, "learning_rate": 7.0909090909090916e-06, "loss": 0.0198, "num_tokens": 2155316.0, "reward": 0.9629546403884888, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9629546403884888, "reward_meter_std": 0.05961780995130539, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05961783230304718, "reward_total_composite_mean": 0.9629546403884888, "reward_total_composite_std": 0.05961780995130539, "reward_total_mean": 0.9629546403884888, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9629546403884888, "rewards/meter/std": 0.05961780995130539, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9629546403884888, "rewards/total_composite/std": 0.05961780995130539, "sampling/importance_sampling_ratio/max": 1.2267403602600098, "sampling/importance_sampling_ratio/mean": 0.9952221512794495, "sampling/importance_sampling_ratio/min": 0.18683788180351257, "sampling/sampling_logp_difference/max": 1.6775139570236206, "sampling/sampling_logp_difference/mean": 0.024385379627346992, "step": 961 }, { "clip_ratio/high_max": 0.015785872004926205, "clip_ratio/high_mean": 0.015785872004926205, "clip_ratio/low_mean": 0.01819333794992417, "clip_ratio/low_min": 0.01819333794992417, "clip_ratio/region_mean": 0.033979209954850376, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.24257362261414528, "epoch": 0.03863919347712576, "frac_reward_zero_std": 0.0, "grad_norm": 8.99122142791748, "learning_rate": 7.087878787878788e-06, "loss": 0.0499, "num_tokens": 2157171.0, "reward": 0.32342612743377686, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.32342612743377686, "reward_meter_std": 0.291739284992218, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.291739284992218, "reward_total_composite_mean": 0.32342612743377686, "reward_total_composite_std": 0.291739284992218, "reward_total_mean": 0.32342612743377686, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.32342612743377686, "rewards/meter/std": 0.291739284992218, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.32342612743377686, "rewards/total_composite/std": 0.291739284992218, "sampling/importance_sampling_ratio/max": 1.8445723056793213, "sampling/importance_sampling_ratio/mean": 1.0009416341781616, "sampling/importance_sampling_ratio/min": 0.1935305893421173, "sampling/sampling_logp_difference/max": 1.642319679260254, "sampling/sampling_logp_difference/mean": 0.04449179768562317, "step": 962 }, { "clip_ratio/high_max": 0.0073644123040139675, "clip_ratio/high_mean": 0.0073644123040139675, "clip_ratio/low_mean": 0.002477022586390376, "clip_ratio/low_min": 0.002477022586390376, "clip_ratio/region_mean": 0.009841434890404344, "completions/clipped_ratio": 0.0, "completions/max_length": 468.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 445.0, "completions/mean_terminated_length": 445.0, "completions/min_length": 435.0, "completions/min_terminated_length": 435.0, "entropy": 0.03673372324556112, "epoch": 0.03867935895891071, "frac_reward_zero_std": 0.0, "grad_norm": 1.0686217546463013, "learning_rate": 7.084848484848485e-06, "loss": 0.0162, "num_tokens": 2162491.0, "reward": 0.3684366047382355, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7291666269302368, "reward_count_adherence_std": 0.01964186504483223, "reward_meter_mean": 0.8052209615707397, "reward_meter_std": 0.37245726585388184, "reward_repeat_penalty_mean": 0.6346794962882996, "reward_repeat_penalty_std": 0.046481478959321976, "reward_std": 0.1665974259376526, "reward_total_composite_mean": 0.3684366047382355, "reward_total_composite_std": 0.16659744083881378, "reward_total_mean": 0.3684366047382355, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7291666269302368, "rewards/count_adherence/std": 0.01964186504483223, "rewards/meter/mean": 0.8052209615707397, "rewards/meter/std": 0.37245726585388184, "rewards/repeat_penalty/mean": 0.6346794962882996, "rewards/repeat_penalty/std": 0.046481478959321976, "rewards/total_composite/mean": 0.3684366047382355, "rewards/total_composite/std": 0.16659744083881378, "sampling/importance_sampling_ratio/max": 1.6414300203323364, "sampling/importance_sampling_ratio/mean": 0.9988970160484314, "sampling/importance_sampling_ratio/min": 0.11253420263528824, "sampling/sampling_logp_difference/max": 2.1844980716705322, "sampling/sampling_logp_difference/mean": 0.010675698518753052, "step": 963 }, { "clip_ratio/high_max": 0.0026147099561057985, "clip_ratio/high_mean": 0.0026147099561057985, "clip_ratio/low_mean": 0.0005319148767739534, "clip_ratio/low_min": 0.0005319148767739534, "clip_ratio/region_mean": 0.003146624832879752, "completions/clipped_ratio": 0.0, "completions/max_length": 261.0, "completions/max_terminated_length": 261.0, "completions/mean_length": 240.75, "completions/mean_terminated_length": 240.75, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.012361971399514005, "epoch": 0.038719524440695666, "frac_reward_zero_std": 0.0, "grad_norm": 2.381939649581909, "learning_rate": 7.081818181818182e-06, "loss": -0.005, "num_tokens": 2165977.0, "reward": 0.5972245931625366, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9859161376953125, "reward_meter_std": 0.0027882950380444527, "reward_repeat_penalty_mean": 0.6057692766189575, "reward_repeat_penalty_std": 0.027196412906050682, "reward_std": 0.026564771309494972, "reward_total_composite_mean": 0.5972245931625366, "reward_total_composite_std": 0.026564769446849823, "reward_total_mean": 0.5972245931625366, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9859161376953125, "rewards/meter/std": 0.0027882950380444527, "rewards/repeat_penalty/mean": 0.6057692766189575, "rewards/repeat_penalty/std": 0.027196412906050682, "rewards/total_composite/mean": 0.5972245931625366, "rewards/total_composite/std": 0.026564769446849823, "sampling/importance_sampling_ratio/max": 1.6210657358169556, "sampling/importance_sampling_ratio/mean": 1.0001258850097656, "sampling/importance_sampling_ratio/min": 0.23969349265098572, "sampling/sampling_logp_difference/max": 1.4283943176269531, "sampling/sampling_logp_difference/mean": 0.004197314847260714, "step": 964 }, { "clip_ratio/high_max": 0.005208333372138441, "clip_ratio/high_mean": 0.005208333372138441, "clip_ratio/low_mean": 0.0036498295376077294, "clip_ratio/low_min": 0.0036498295376077294, "clip_ratio/region_mean": 0.00885816290974617, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 71.875, "completions/mean_terminated_length": 71.875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.03642029734328389, "epoch": 0.03875968992248062, "frac_reward_zero_std": 0.0, "grad_norm": 6.1013593673706055, "learning_rate": 7.07878787878788e-06, "loss": 0.0083, "num_tokens": 2168080.0, "reward": 0.9859936833381653, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9859936833381653, "reward_meter_std": 0.00813794881105423, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008137939497828484, "reward_total_composite_mean": 0.9859936833381653, "reward_total_composite_std": 0.00813794881105423, "reward_total_mean": 0.9859936833381653, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9859936833381653, "rewards/meter/std": 0.00813794881105423, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9859936833381653, "rewards/total_composite/std": 0.00813794881105423, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001380443572998, "sampling/importance_sampling_ratio/min": 0.6168099045753479, "sampling/sampling_logp_difference/max": 0.8346943855285645, "sampling/sampling_logp_difference/mean": 0.008848754689097404, "step": 965 }, { "clip_ratio/high_max": 0.048127192771062255, "clip_ratio/high_mean": 0.048127192771062255, "clip_ratio/low_mean": 0.005769230891019106, "clip_ratio/low_min": 0.005769230891019106, "clip_ratio/region_mean": 0.05389642366208136, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 64.75, "completions/mean_terminated_length": 64.75, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.18688618578016758, "epoch": 0.038799855404265574, "frac_reward_zero_std": 0.0, "grad_norm": 10.40634822845459, "learning_rate": 7.075757575757576e-06, "loss": 0.0052, "num_tokens": 2169886.0, "reward": 0.7862817645072937, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7862817645072937, "reward_meter_std": 0.3110516667366028, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3110516369342804, "reward_total_composite_mean": 0.7862817645072937, "reward_total_composite_std": 0.3110516667366028, "reward_total_mean": 0.7862817645072937, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7862817645072937, "rewards/meter/std": 0.3110516667366028, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7862817645072937, "rewards/total_composite/std": 0.3110516667366028, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0035208463668823, "sampling/importance_sampling_ratio/min": 0.3482150137424469, "sampling/sampling_logp_difference/max": 1.9868323802947998, "sampling/sampling_logp_difference/mean": 0.05011482164263725, "step": 966 }, { "clip_ratio/high_max": 0.004360174119938165, "clip_ratio/high_mean": 0.004360174119938165, "clip_ratio/low_mean": 0.005190499185118824, "clip_ratio/low_min": 0.005190499185118824, "clip_ratio/region_mean": 0.00955067330505699, "completions/clipped_ratio": 0.0, "completions/max_length": 206.0, "completions/max_terminated_length": 206.0, "completions/mean_length": 194.375, "completions/mean_terminated_length": 194.375, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.03619017521850765, "epoch": 0.03884002088605053, "frac_reward_zero_std": 0.0, "grad_norm": 2.4970195293426514, "learning_rate": 7.072727272727273e-06, "loss": -0.0072, "num_tokens": 2172905.0, "reward": 0.39438170194625854, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5914201736450195, "reward_meter_std": 0.4835823178291321, "reward_repeat_penalty_mean": 0.6805555820465088, "reward_repeat_penalty_std": 0.03928370773792267, "reward_std": 0.32224807143211365, "reward_total_composite_mean": 0.39438170194625854, "reward_total_composite_std": 0.32224810123443604, "reward_total_mean": 0.39438170194625854, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5914201736450195, "rewards/meter/std": 0.4835823178291321, "rewards/repeat_penalty/mean": 0.6805555820465088, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.39438170194625854, "rewards/total_composite/std": 0.32224810123443604, "sampling/importance_sampling_ratio/max": 1.642437219619751, "sampling/importance_sampling_ratio/mean": 1.000235676765442, "sampling/importance_sampling_ratio/min": 0.1496105194091797, "sampling/sampling_logp_difference/max": 1.8997198343276978, "sampling/sampling_logp_difference/mean": 0.008972376585006714, "step": 967 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.03520799265243113, "clip_ratio/low_min": 0.03520799265243113, "clip_ratio/region_mean": 0.03520799265243113, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 60.5, "completions/mean_terminated_length": 60.5, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.0875935684889555, "epoch": 0.03888018636783548, "frac_reward_zero_std": 0.0, "grad_norm": 5.905569076538086, "learning_rate": 7.06969696969697e-06, "loss": -0.0085, "num_tokens": 2174629.0, "reward": 0.1817864626646042, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.1817864626646042, "reward_meter_std": 0.293150931596756, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.293150931596756, "reward_total_composite_mean": 0.1817864626646042, "reward_total_composite_std": 0.293150931596756, "reward_total_mean": 0.1817864626646042, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.1817864626646042, "rewards/meter/std": 0.293150931596756, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.1817864626646042, "rewards/total_composite/std": 0.293150931596756, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0088768005371094, "sampling/importance_sampling_ratio/min": 0.4889802932739258, "sampling/sampling_logp_difference/max": 0.7234911918640137, "sampling/sampling_logp_difference/mean": 0.026289377361536026, "step": 968 }, { "clip_ratio/high_max": 0.01694070769008249, "clip_ratio/high_mean": 0.01694070769008249, "clip_ratio/low_mean": 0.009395235683768988, "clip_ratio/low_min": 0.009395235683768988, "clip_ratio/region_mean": 0.026335943373851478, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 93.75, "completions/mean_terminated_length": 93.75, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.07108194520696998, "epoch": 0.038920351849620435, "frac_reward_zero_std": 0.0, "grad_norm": 3.71621036529541, "learning_rate": 7.066666666666667e-06, "loss": -0.0153, "num_tokens": 2176683.0, "reward": 0.4796490967273712, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5728106498718262, "reward_meter_std": 0.2251601219177246, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.12817399203777313, "reward_std": 0.21056686341762543, "reward_total_composite_mean": 0.4796490967273712, "reward_total_composite_std": 0.21056687831878662, "reward_total_mean": 0.4796490967273712, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5728106498718262, "rewards/meter/std": 0.2251601219177246, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.4796490967273712, "rewards/total_composite/std": 0.21056687831878662, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9970004558563232, "sampling/importance_sampling_ratio/min": 0.18678979575634003, "sampling/sampling_logp_difference/max": 1.6777713298797607, "sampling/sampling_logp_difference/mean": 0.024638189002871513, "step": 969 }, { "clip_ratio/high_max": 0.006389971007592976, "clip_ratio/high_mean": 0.006389971007592976, "clip_ratio/low_mean": 0.011199243483133614, "clip_ratio/low_min": 0.011199243483133614, "clip_ratio/region_mean": 0.01758921449072659, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 78.75, "completions/mean_terminated_length": 78.75, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.09748060069978237, "epoch": 0.03896051733140539, "frac_reward_zero_std": 0.0, "grad_norm": 2.2865495681762695, "learning_rate": 7.063636363636365e-06, "loss": -0.0005, "num_tokens": 2178721.0, "reward": 0.9977834224700928, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977834224700928, "reward_meter_std": 0.00039716452010907233, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0003971673722844571, "reward_total_composite_mean": 0.9977834224700928, "reward_total_composite_std": 0.00039716452010907233, "reward_total_mean": 0.9977834224700928, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977834224700928, "rewards/meter/std": 0.00039716452010907233, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977834224700928, "rewards/total_composite/std": 0.00039716452010907233, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0018593072891235, "sampling/importance_sampling_ratio/min": 0.4526699483394623, "sampling/sampling_logp_difference/max": 0.7925920486450195, "sampling/sampling_logp_difference/mean": 0.019730689004063606, "step": 970 }, { "clip_ratio/high_max": 0.010603155242279172, "clip_ratio/high_mean": 0.010603155242279172, "clip_ratio/low_mean": 0.002027027076110244, "clip_ratio/low_min": 0.002027027076110244, "clip_ratio/region_mean": 0.012630182318389416, "completions/clipped_ratio": 0.0, "completions/max_length": 192.0, "completions/max_terminated_length": 192.0, "completions/mean_length": 189.0, "completions/mean_terminated_length": 189.0, "completions/min_length": 183.0, "completions/min_terminated_length": 183.0, "entropy": 0.06829563481733203, "epoch": 0.03900068281319034, "frac_reward_zero_std": 0.0, "grad_norm": 1.4529743194580078, "learning_rate": 7.060606060606061e-06, "loss": -0.0062, "num_tokens": 2181793.0, "reward": 0.6518542766571045, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985735416412354, "reward_meter_std": 0.0002435932110529393, "reward_repeat_penalty_mean": 0.6527777910232544, "reward_repeat_penalty_std": 0.13849149644374847, "reward_std": 0.13831683993339539, "reward_total_composite_mean": 0.6518542766571045, "reward_total_composite_std": 0.13831683993339539, "reward_total_mean": 0.6518542766571045, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985735416412354, "rewards/meter/std": 0.0002435932110529393, "rewards/repeat_penalty/mean": 0.6527777910232544, "rewards/repeat_penalty/std": 0.13849149644374847, "rewards/total_composite/mean": 0.6518542766571045, "rewards/total_composite/std": 0.13831683993339539, "sampling/importance_sampling_ratio/max": 1.8489798307418823, "sampling/importance_sampling_ratio/mean": 0.998883068561554, "sampling/importance_sampling_ratio/min": 0.0026406734250485897, "sampling/sampling_logp_difference/max": 5.936721324920654, "sampling/sampling_logp_difference/mean": 0.016793714836239815, "step": 971 }, { "clip_ratio/high_max": 0.0037591225700452924, "clip_ratio/high_mean": 0.0037591225700452924, "clip_ratio/low_mean": 0.006205761310411617, "clip_ratio/low_min": 0.006205761310411617, "clip_ratio/region_mean": 0.00996488388045691, "completions/clipped_ratio": 0.0, "completions/max_length": 273.0, "completions/max_terminated_length": 273.0, "completions/mean_length": 263.5, "completions/mean_terminated_length": 263.5, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.05923895724117756, "epoch": 0.0390408482949753, "frac_reward_zero_std": 0.0, "grad_norm": 1.3656582832336426, "learning_rate": 7.057575757575759e-06, "loss": -0.01, "num_tokens": 2185373.0, "reward": 0.19842997193336487, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.859375, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.6416140794754028, "reward_meter_std": 0.4567407965660095, "reward_repeat_penalty_mean": 0.4212246239185333, "reward_repeat_penalty_std": 0.21046686172485352, "reward_std": 0.19151146709918976, "reward_total_composite_mean": 0.19842997193336487, "reward_total_composite_std": 0.19151148200035095, "reward_total_mean": 0.19842997193336487, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.859375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.6416140794754028, "rewards/meter/std": 0.4567407965660095, "rewards/repeat_penalty/mean": 0.4212246239185333, "rewards/repeat_penalty/std": 0.21046686172485352, "rewards/total_composite/mean": 0.19842997193336487, "rewards/total_composite/std": 0.19151148200035095, "sampling/importance_sampling_ratio/max": 1.6406102180480957, "sampling/importance_sampling_ratio/mean": 1.0005074739456177, "sampling/importance_sampling_ratio/min": 0.0007901470526121557, "sampling/sampling_logp_difference/max": 7.143291473388672, "sampling/sampling_logp_difference/mean": 0.017121732234954834, "step": 972 }, { "clip_ratio/high_max": 0.010416667209938169, "clip_ratio/high_mean": 0.010416667209938169, "clip_ratio/low_mean": 0.008165820967406034, "clip_ratio/low_min": 0.008165820967406034, "clip_ratio/region_mean": 0.018582488177344203, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 60.75, "completions/mean_terminated_length": 60.75, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.09659606032073498, "epoch": 0.03908101377676025, "frac_reward_zero_std": 0.0, "grad_norm": 13.225342750549316, "learning_rate": 7.054545454545455e-06, "loss": 0.0171, "num_tokens": 2187067.0, "reward": 0.5853145122528076, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5853145122528076, "reward_meter_std": 0.1321393847465515, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.13213936984539032, "reward_total_composite_mean": 0.5853145122528076, "reward_total_composite_std": 0.1321393847465515, "reward_total_mean": 0.5853145122528076, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5853145122528076, "rewards/meter/std": 0.1321393847465515, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5853145122528076, "rewards/total_composite/std": 0.1321393847465515, "sampling/importance_sampling_ratio/max": 1.8551063537597656, "sampling/importance_sampling_ratio/mean": 0.9999557733535767, "sampling/importance_sampling_ratio/min": 0.2996131479740143, "sampling/sampling_logp_difference/max": 1.2052631378173828, "sampling/sampling_logp_difference/mean": 0.022791719064116478, "step": 973 }, { "clip_ratio/high_max": 0.004807692370377481, "clip_ratio/high_mean": 0.004807692370377481, "clip_ratio/low_mean": 0.006329114083200693, "clip_ratio/low_min": 0.006329114083200693, "clip_ratio/region_mean": 0.011136806453578174, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 78.875, "completions/mean_terminated_length": 78.875, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.08664691727608442, "epoch": 0.039121179258545205, "frac_reward_zero_std": 0.0, "grad_norm": 2.125204563140869, "learning_rate": 7.0515151515151525e-06, "loss": 0.0027, "num_tokens": 2188986.0, "reward": 0.9978776574134827, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978776574134827, "reward_meter_std": 0.0004215049266349524, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004215273365844041, "reward_total_composite_mean": 0.9978776574134827, "reward_total_composite_std": 0.0004215049266349524, "reward_total_mean": 0.9978776574134827, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978776574134827, "rewards/meter/std": 0.0004215049266349524, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978776574134827, "rewards/total_composite/std": 0.0004215049266349524, "sampling/importance_sampling_ratio/max": 1.3348405361175537, "sampling/importance_sampling_ratio/mean": 1.0038025379180908, "sampling/importance_sampling_ratio/min": 0.503719687461853, "sampling/sampling_logp_difference/max": 0.6857354640960693, "sampling/sampling_logp_difference/mean": 0.013205158524215221, "step": 974 }, { "clip_ratio/high_max": 0.019024619832634926, "clip_ratio/high_mean": 0.019024619832634926, "clip_ratio/low_mean": 0.003676470718346536, "clip_ratio/low_min": 0.003676470718346536, "clip_ratio/region_mean": 0.022701090550981462, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 100.875, "completions/mean_terminated_length": 100.875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.08865811070427299, "epoch": 0.03916134474033016, "frac_reward_zero_std": 0.0, "grad_norm": 2.8000571727752686, "learning_rate": 7.048484848484849e-06, "loss": 0.0108, "num_tokens": 2191193.0, "reward": 0.921911358833313, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996698796749115, "reward_meter_std": 0.0009329493041150272, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10283298790454865, "reward_total_composite_mean": 0.921911358833313, "reward_total_composite_std": 0.10283299535512924, "reward_total_mean": 0.921911358833313, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996698796749115, "rewards/meter/std": 0.0009329493041150272, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.921911358833313, "rewards/total_composite/std": 0.10283299535512924, "sampling/importance_sampling_ratio/max": 1.8002057075500488, "sampling/importance_sampling_ratio/mean": 0.9971323609352112, "sampling/importance_sampling_ratio/min": 0.1370241641998291, "sampling/sampling_logp_difference/max": 1.987597942352295, "sampling/sampling_logp_difference/mean": 0.027565594762563705, "step": 975 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.006250000325962901, "clip_ratio/low_min": 0.006250000325962901, "clip_ratio/region_mean": 0.006250000325962901, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 60.25, "completions/mean_terminated_length": 60.25, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.029942457331344485, "epoch": 0.03920151022211511, "frac_reward_zero_std": 0.0, "grad_norm": 8.02316665649414, "learning_rate": 7.045454545454546e-06, "loss": -0.0053, "num_tokens": 2193051.0, "reward": 0.6566718816757202, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9850078821182251, "reward_meter_std": 0.0014023992698639631, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009349363390356302, "reward_total_composite_mean": 0.6566718816757202, "reward_total_composite_std": 0.0009349193423986435, "reward_total_mean": 0.6566718816757202, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9850078821182251, "rewards/meter/std": 0.0014023992698639631, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6566718816757202, "rewards/total_composite/std": 0.0009349193423986435, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0041868686676025, "sampling/importance_sampling_ratio/min": 0.4981825351715088, "sampling/sampling_logp_difference/max": 0.7054405212402344, "sampling/sampling_logp_difference/mean": 0.009783357381820679, "step": 976 }, { "clip_ratio/high_max": 0.007735339575447142, "clip_ratio/high_mean": 0.007735339575447142, "clip_ratio/low_mean": 0.006289557088166475, "clip_ratio/low_min": 0.006289557088166475, "clip_ratio/region_mean": 0.014024896663613617, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 79.625, "completions/mean_terminated_length": 79.625, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.09024440124630928, "epoch": 0.03924167570390007, "frac_reward_zero_std": 0.0, "grad_norm": 1.635718584060669, "learning_rate": 7.0424242424242426e-06, "loss": -0.0065, "num_tokens": 2194984.0, "reward": 0.9982393980026245, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982393980026245, "reward_meter_std": 0.0004721701261587441, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00047218104009516537, "reward_total_composite_mean": 0.9982393980026245, "reward_total_composite_std": 0.0004721701261587441, "reward_total_mean": 0.9982393980026245, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982393980026245, "rewards/meter/std": 0.0004721701261587441, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982393980026245, "rewards/total_composite/std": 0.0004721701261587441, "sampling/importance_sampling_ratio/max": 1.298281192779541, "sampling/importance_sampling_ratio/mean": 1.0005873441696167, "sampling/importance_sampling_ratio/min": 0.5519744157791138, "sampling/sampling_logp_difference/max": 0.5942535400390625, "sampling/sampling_logp_difference/mean": 0.013445981778204441, "step": 977 }, { "clip_ratio/high_max": 0.008984935469925404, "clip_ratio/high_mean": 0.008984935469925404, "clip_ratio/low_mean": 0.0029248768696561456, "clip_ratio/low_min": 0.0029248768696561456, "clip_ratio/region_mean": 0.01190981233958155, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 84.625, "completions/mean_terminated_length": 84.625, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.06632238999009132, "epoch": 0.03928184118568502, "frac_reward_zero_std": 0.0, "grad_norm": 3.430257558822632, "learning_rate": 7.039393939393941e-06, "loss": 0.0049, "num_tokens": 2196933.0, "reward": 0.7949336171150208, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99366694688797, "reward_meter_std": 0.0010994295589625835, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008795478497631848, "reward_total_composite_mean": 0.7949336171150208, "reward_total_composite_std": 0.0008795479079708457, "reward_total_mean": 0.7949336171150208, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99366694688797, "rewards/meter/std": 0.0010994295589625835, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7949336171150208, "rewards/total_composite/std": 0.0008795479079708457, "sampling/importance_sampling_ratio/max": 1.2917805910110474, "sampling/importance_sampling_ratio/mean": 0.9972406029701233, "sampling/importance_sampling_ratio/min": 0.23198257386684418, "sampling/sampling_logp_difference/max": 1.4610930681228638, "sampling/sampling_logp_difference/mean": 0.016848498955368996, "step": 978 }, { "clip_ratio/high_max": 0.020312500651925802, "clip_ratio/high_mean": 0.020312500651925802, "clip_ratio/low_mean": 0.016670140204951167, "clip_ratio/low_min": 0.016670140204951167, "clip_ratio/region_mean": 0.03698264085687697, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 60.375, "completions/mean_terminated_length": 60.375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.1247731251642108, "epoch": 0.039322006667469975, "frac_reward_zero_std": 0.0, "grad_norm": 6.046172618865967, "learning_rate": 7.036363636363637e-06, "loss": -0.0028, "num_tokens": 2198744.0, "reward": 0.6340106725692749, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6340106725692749, "reward_meter_std": 0.20943842828273773, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.20943844318389893, "reward_total_composite_mean": 0.6340106725692749, "reward_total_composite_std": 0.20943842828273773, "reward_total_mean": 0.6340106725692749, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6340106725692749, "rewards/meter/std": 0.20943842828273773, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6340106725692749, "rewards/total_composite/std": 0.20943842828273773, "sampling/importance_sampling_ratio/max": 1.586508870124817, "sampling/importance_sampling_ratio/mean": 0.9989505410194397, "sampling/importance_sampling_ratio/min": 0.27730610966682434, "sampling/sampling_logp_difference/max": 1.2826333045959473, "sampling/sampling_logp_difference/mean": 0.030584923923015594, "step": 979 }, { "clip_ratio/high_max": 0.0010683761211112142, "clip_ratio/high_mean": 0.0010683761211112142, "clip_ratio/low_mean": 0.005235925782471895, "clip_ratio/low_min": 0.005235925782471895, "clip_ratio/region_mean": 0.006304301903583109, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 117.625, "completions/mean_terminated_length": 117.625, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.0817764438688755, "epoch": 0.03936217214925493, "frac_reward_zero_std": 0.0, "grad_norm": 1.3390476703643799, "learning_rate": 7.033333333333334e-06, "loss": -0.0001, "num_tokens": 2201037.0, "reward": 0.7989562153816223, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986953139305115, "reward_meter_std": 0.00018651132995728403, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00014920474495738745, "reward_total_composite_mean": 0.7989562153816223, "reward_total_composite_std": 0.00014920807734597474, "reward_total_mean": 0.7989562153816223, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986953139305115, "rewards/meter/std": 0.00018651132995728403, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7989562153816223, "rewards/total_composite/std": 0.00014920807734597474, "sampling/importance_sampling_ratio/max": 1.392466425895691, "sampling/importance_sampling_ratio/mean": 1.001077651977539, "sampling/importance_sampling_ratio/min": 0.2592008113861084, "sampling/sampling_logp_difference/max": 1.3501521348953247, "sampling/sampling_logp_difference/mean": 0.012999390251934528, "step": 980 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.012774199014529586, "epoch": 0.03940233763103988, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.030303030303031e-06, "loss": 0.0, "num_tokens": 2202789.0, "reward": 0.6581420302391052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9872130751609802, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.6581420302391052, "reward_total_composite_std": 0.0, "reward_total_mean": 0.6581420302391052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9872130751609802, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6581420302391052, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0427228212356567, "sampling/importance_sampling_ratio/mean": 1.001099944114685, "sampling/importance_sampling_ratio/min": 0.9491093754768372, "sampling/sampling_logp_difference/max": 0.05223120003938675, "sampling/sampling_logp_difference/mean": 0.0014747538371011615, "step": 981 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 52.0, "completions/max_terminated_length": 52.0, "completions/mean_length": 52.0, "completions/mean_terminated_length": 52.0, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.04045943822711706, "epoch": 0.03944250311282484, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.027272727272728e-06, "loss": 0.0, "num_tokens": 2204653.0, "reward": 0.961655855178833, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.961655855178833, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.961655855178833, "reward_total_composite_std": 0.0, "reward_total_mean": 0.961655855178833, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.961655855178833, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.961655855178833, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.1538965702056885, "sampling/importance_sampling_ratio/mean": 1.0008008480072021, "sampling/importance_sampling_ratio/min": 0.44697439670562744, "sampling/sampling_logp_difference/max": 0.8052540421485901, "sampling/sampling_logp_difference/mean": 0.007805653847754002, "step": 982 }, { "clip_ratio/high_max": 0.0030487803742289543, "clip_ratio/high_mean": 0.0030487803742289543, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/region_mean": 0.004611280397512019, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 80.5, "completions/mean_terminated_length": 80.5, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.07731548231095076, "epoch": 0.03948266859460979, "frac_reward_zero_std": 0.0, "grad_norm": 2.655090093612671, "learning_rate": 7.024242424242424e-06, "loss": -0.0039, "num_tokens": 2206569.0, "reward": 0.998711347579956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998711347579956, "reward_meter_std": 0.0004273413505870849, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000427335879066959, "reward_total_composite_mean": 0.998711347579956, "reward_total_composite_std": 0.0004273413505870849, "reward_total_mean": 0.998711347579956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998711347579956, "rewards/meter/std": 0.0004273413505870849, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998711347579956, "rewards/total_composite/std": 0.0004273413505870849, "sampling/importance_sampling_ratio/max": 1.6649136543273926, "sampling/importance_sampling_ratio/mean": 1.0012972354888916, "sampling/importance_sampling_ratio/min": 0.40555310249328613, "sampling/sampling_logp_difference/max": 0.902503490447998, "sampling/sampling_logp_difference/mean": 0.01396770216524601, "step": 983 }, { "clip_ratio/high_max": 0.007938507944345474, "clip_ratio/high_mean": 0.007938507944345474, "clip_ratio/low_mean": 0.016003023833036423, "clip_ratio/low_min": 0.016003023833036423, "clip_ratio/region_mean": 0.023941531777381897, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 31.375, "completions/mean_terminated_length": 31.375, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.08878939691931009, "epoch": 0.039522834076394744, "frac_reward_zero_std": 0.0, "grad_norm": 4.302158832550049, "learning_rate": 7.021212121212122e-06, "loss": -0.0007, "num_tokens": 2208124.0, "reward": 0.9924632906913757, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9924632906913757, "reward_meter_std": 0.0021250685676932335, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0021250564604997635, "reward_total_composite_mean": 0.9924632906913757, "reward_total_composite_std": 0.0021250685676932335, "reward_total_mean": 0.9924632906913757, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9924632906913757, "rewards/meter/std": 0.0021250685676932335, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924632906913757, "rewards/total_composite/std": 0.0021250685676932335, "sampling/importance_sampling_ratio/max": 1.7839521169662476, "sampling/importance_sampling_ratio/mean": 1.0098501443862915, "sampling/importance_sampling_ratio/min": 0.7517499923706055, "sampling/sampling_logp_difference/max": 0.5788311958312988, "sampling/sampling_logp_difference/mean": 0.014777499251067638, "step": 984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.020491802133619785, "clip_ratio/low_min": 0.020491802133619785, "clip_ratio/region_mean": 0.020491802133619785, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.25, "completions/mean_terminated_length": 61.25, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.04339231993071735, "epoch": 0.0395629995581797, "frac_reward_zero_std": 0.0, "grad_norm": 8.724225997924805, "learning_rate": 7.018181818181818e-06, "loss": -0.0119, "num_tokens": 2209934.0, "reward": 0.6991585493087769, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9870957136154175, "reward_meter_std": 0.00033180107129737735, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_std": 0.11601236462593079, "reward_total_composite_mean": 0.6991585493087769, "reward_total_composite_std": 0.11601238697767258, "reward_total_mean": 0.6991585493087769, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9870957136154175, "rewards/meter/std": 0.00033180107129737735, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.6991585493087769, "rewards/total_composite/std": 0.11601238697767258, "sampling/importance_sampling_ratio/max": 1.6728845834732056, "sampling/importance_sampling_ratio/mean": 1.0018260478973389, "sampling/importance_sampling_ratio/min": 0.3336217403411865, "sampling/sampling_logp_difference/max": 1.0977474451065063, "sampling/sampling_logp_difference/mean": 0.012023198418319225, "step": 985 }, { "clip_ratio/high_max": 0.037494323682039976, "clip_ratio/high_mean": 0.037494323682039976, "clip_ratio/low_mean": 0.003989361692219973, "clip_ratio/low_min": 0.003989361692219973, "clip_ratio/region_mean": 0.04148368537425995, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 99.25, "completions/mean_terminated_length": 99.25, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.17434827517718077, "epoch": 0.03960316503996465, "frac_reward_zero_std": 0.0, "grad_norm": 10.352254867553711, "learning_rate": 7.015151515151516e-06, "loss": -0.0012, "num_tokens": 2212176.0, "reward": 0.9695063829421997, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9944287538528442, "reward_meter_std": 0.0026359334588050842, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06953759491443634, "reward_total_composite_mean": 0.9695063829421997, "reward_total_composite_std": 0.06953759491443634, "reward_total_mean": 0.9695063829421997, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9944287538528442, "rewards/meter/std": 0.0026359334588050842, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9695063829421997, "rewards/total_composite/std": 0.06953759491443634, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007062315940857, "sampling/importance_sampling_ratio/min": 0.11121489107608795, "sampling/sampling_logp_difference/max": 2.196290969848633, "sampling/sampling_logp_difference/mean": 0.044116295874118805, "step": 986 }, { "clip_ratio/high_max": 0.0040023052133619785, "clip_ratio/high_mean": 0.0040023052133619785, "clip_ratio/low_mean": 0.022379032103344798, "clip_ratio/low_min": 0.022379032103344798, "clip_ratio/region_mean": 0.026381337316706777, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 61.125, "completions/mean_terminated_length": 61.125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.08241967670619488, "epoch": 0.039643330521749606, "frac_reward_zero_std": 0.0, "grad_norm": 4.8408732414245605, "learning_rate": 7.0121212121212126e-06, "loss": -0.0152, "num_tokens": 2213841.0, "reward": 0.7787973880767822, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9843782782554626, "reward_meter_std": 0.00654714647680521, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.16700218617916107, "reward_total_composite_mean": 0.7787973880767822, "reward_total_composite_std": 0.16700218617916107, "reward_total_mean": 0.7787973880767822, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9843782782554626, "rewards/meter/std": 0.00654714647680521, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7787973880767822, "rewards/total_composite/std": 0.16700218617916107, "sampling/importance_sampling_ratio/max": 1.5299021005630493, "sampling/importance_sampling_ratio/mean": 0.9993690252304077, "sampling/importance_sampling_ratio/min": 0.22009159624576569, "sampling/sampling_logp_difference/max": 1.5137114524841309, "sampling/sampling_logp_difference/mean": 0.02119199000298977, "step": 987 }, { "clip_ratio/high_max": 0.007151253987103701, "clip_ratio/high_mean": 0.007151253987103701, "clip_ratio/low_mean": 0.019569998257793486, "clip_ratio/low_min": 0.019569998257793486, "clip_ratio/region_mean": 0.026721252244897187, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 89.0, "completions/mean_terminated_length": 89.0, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.10643400624394417, "epoch": 0.03968349600353456, "frac_reward_zero_std": 0.0, "grad_norm": 3.5085859298706055, "learning_rate": 7.00909090909091e-06, "loss": 0.0136, "num_tokens": 2215857.0, "reward": 0.8443044424057007, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9934116005897522, "reward_meter_std": 0.0031344336457550526, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09088362008333206, "reward_total_composite_mean": 0.8443044424057007, "reward_total_composite_std": 0.09088363498449326, "reward_total_mean": 0.8443044424057007, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9934116005897522, "rewards/meter/std": 0.0031344336457550526, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8443044424057007, "rewards/total_composite/std": 0.09088363498449326, "sampling/importance_sampling_ratio/max": 1.6077567338943481, "sampling/importance_sampling_ratio/mean": 1.0044476985931396, "sampling/importance_sampling_ratio/min": 0.24122609198093414, "sampling/sampling_logp_difference/max": 1.422020673751831, "sampling/sampling_logp_difference/mean": 0.023283695802092552, "step": 988 }, { "clip_ratio/high_max": 0.014212056528776884, "clip_ratio/high_mean": 0.014212056528776884, "clip_ratio/low_mean": 0.01203381922096014, "clip_ratio/low_min": 0.01203381922096014, "clip_ratio/region_mean": 0.026245875749737024, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.625, "completions/mean_terminated_length": 61.625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.14261625241488218, "epoch": 0.039723661485319514, "frac_reward_zero_std": 0.0, "grad_norm": 4.891083717346191, "learning_rate": 7.006060606060606e-06, "loss": 0.0136, "num_tokens": 2217566.0, "reward": 0.9817750453948975, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9817750453948975, "reward_meter_std": 0.007981428876519203, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00798142608255148, "reward_total_composite_mean": 0.9817750453948975, "reward_total_composite_std": 0.007981428876519203, "reward_total_mean": 0.9817750453948975, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9817750453948975, "rewards/meter/std": 0.007981428876519203, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9817750453948975, "rewards/total_composite/std": 0.007981428876519203, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0027326345443726, "sampling/importance_sampling_ratio/min": 0.27682268619537354, "sampling/sampling_logp_difference/max": 1.2843780517578125, "sampling/sampling_logp_difference/mean": 0.02616567350924015, "step": 989 }, { "clip_ratio/high_max": 0.022955394815653563, "clip_ratio/high_mean": 0.022955394815653563, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/region_mean": 0.027192682959139347, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 60.0, "completions/mean_terminated_length": 60.0, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.09698301088064909, "epoch": 0.03976382696710447, "frac_reward_zero_std": 0.0, "grad_norm": 5.835471153259277, "learning_rate": 7.0030303030303035e-06, "loss": 0.0012, "num_tokens": 2219326.0, "reward": 0.9495816826820374, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9907355308532715, "reward_meter_std": 0.002091656206175685, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11764255911111832, "reward_total_composite_mean": 0.9495816826820374, "reward_total_composite_std": 0.11764256656169891, "reward_total_mean": 0.9495816826820374, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9907355308532715, "rewards/meter/std": 0.002091656206175685, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9495816826820374, "rewards/total_composite/std": 0.11764256656169891, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0002789497375488, "sampling/importance_sampling_ratio/min": 0.2323766052722931, "sampling/sampling_logp_difference/max": 1.4593958854675293, "sampling/sampling_logp_difference/mean": 0.024150043725967407, "step": 990 }, { "clip_ratio/high_max": 0.008049219497479498, "clip_ratio/high_mean": 0.008049219497479498, "clip_ratio/low_mean": 0.0006097560981288552, "clip_ratio/low_min": 0.0006097560981288552, "clip_ratio/region_mean": 0.008658975595608354, "completions/clipped_ratio": 0.0, "completions/max_length": 411.0, "completions/max_terminated_length": 411.0, "completions/mean_length": 407.5, "completions/mean_terminated_length": 407.5, "completions/min_length": 394.0, "completions/min_terminated_length": 394.0, "entropy": 0.027644895017147064, "epoch": 0.03980399244888942, "frac_reward_zero_std": 0.0, "grad_norm": 1.7695684432983398, "learning_rate": 7e-06, "loss": 0.0039, "num_tokens": 2224258.0, "reward": 0.3326081335544586, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9920412302017212, "reward_meter_std": 0.003196289064362645, "reward_repeat_penalty_mean": 0.46064817905426025, "reward_repeat_penalty_std": 0.15420734882354736, "reward_std": 0.11179134249687195, "reward_total_composite_mean": 0.3326081335544586, "reward_total_composite_std": 0.11179135739803314, "reward_total_mean": 0.3326081335544586, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9920412302017212, "rewards/meter/std": 0.003196289064362645, "rewards/repeat_penalty/mean": 0.46064817905426025, "rewards/repeat_penalty/std": 0.15420734882354736, "rewards/total_composite/mean": 0.3326081335544586, "rewards/total_composite/std": 0.11179135739803314, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9987042546272278, "sampling/importance_sampling_ratio/min": 0.030707810074090958, "sampling/sampling_logp_difference/max": 3.4832382202148438, "sampling/sampling_logp_difference/mean": 0.01644131913781166, "step": 991 }, { "clip_ratio/high_max": 0.01013427460566163, "clip_ratio/high_mean": 0.01013427460566163, "clip_ratio/low_mean": 0.0047512390883639455, "clip_ratio/low_min": 0.0047512390883639455, "clip_ratio/region_mean": 0.014885513694025576, "completions/clipped_ratio": 0.0, "completions/max_length": 428.0, "completions/max_terminated_length": 428.0, "completions/mean_length": 419.75, "completions/mean_terminated_length": 419.75, "completions/min_length": 409.0, "completions/min_terminated_length": 409.0, "entropy": 0.07821797719225287, "epoch": 0.039844157930674376, "frac_reward_zero_std": 0.0, "grad_norm": 1.6267327070236206, "learning_rate": 6.996969696969698e-06, "loss": 0.0087, "num_tokens": 2229368.0, "reward": 0.4672620892524719, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7954545021057129, "reward_count_adherence_std": 0.04208271950483322, "reward_meter_mean": 0.9450836181640625, "reward_meter_std": 0.07978925108909607, "reward_repeat_penalty_mean": 0.62321937084198, "reward_repeat_penalty_std": 0.04523347690701485, "reward_std": 0.04408339038491249, "reward_total_composite_mean": 0.4672620892524719, "reward_total_composite_std": 0.044083379209041595, "reward_total_mean": 0.4672620892524719, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7954545021057129, "rewards/count_adherence/std": 0.04208271950483322, "rewards/meter/mean": 0.9450836181640625, "rewards/meter/std": 0.07978925108909607, "rewards/repeat_penalty/mean": 0.62321937084198, "rewards/repeat_penalty/std": 0.04523347690701485, "rewards/total_composite/mean": 0.4672620892524719, "rewards/total_composite/std": 0.044083379209041595, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000062108039856, "sampling/importance_sampling_ratio/min": 0.0658869743347168, "sampling/sampling_logp_difference/max": 2.7198145389556885, "sampling/sampling_logp_difference/mean": 0.01929199881851673, "step": 992 }, { "clip_ratio/high_max": 0.021799394860863686, "clip_ratio/high_mean": 0.021799394860863686, "clip_ratio/low_mean": 0.01408805197570473, "clip_ratio/low_min": 0.01408805197570473, "clip_ratio/region_mean": 0.035887446836568415, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 63.125, "completions/mean_terminated_length": 63.125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.1402796907350421, "epoch": 0.03988432341245933, "frac_reward_zero_std": 0.0, "grad_norm": 5.988306522369385, "learning_rate": 6.993939393939394e-06, "loss": -0.0038, "num_tokens": 2231169.0, "reward": 0.6225343942642212, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6225343942642212, "reward_meter_std": 0.3685222268104553, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3685222268104553, "reward_total_composite_mean": 0.6225343942642212, "reward_total_composite_std": 0.3685222268104553, "reward_total_mean": 0.6225343942642212, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6225343942642212, "rewards/meter/std": 0.3685222268104553, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6225343942642212, "rewards/total_composite/std": 0.3685222268104553, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0002118349075317, "sampling/importance_sampling_ratio/min": 0.3208072781562805, "sampling/sampling_logp_difference/max": 1.1369147300720215, "sampling/sampling_logp_difference/mean": 0.02788170613348484, "step": 993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.039924488894244284, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 6.990909090909092e-06, "loss": 0.0, "num_tokens": 2232881.0, "reward": 0.44484955072402954, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8088235855102539, "reward_count_adherence_std": 0.027230001986026764, "reward_meter_mean": 0.9936738014221191, "reward_meter_std": 0.003524529282003641, "reward_repeat_penalty_mean": 0.5536096096038818, "reward_repeat_penalty_std": 0.012604749761521816, "reward_std": 0.015336750075221062, "reward_total_composite_mean": 0.44484955072402954, "reward_total_composite_std": 0.015336754731833935, "reward_total_mean": 0.44484955072402954, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8088235855102539, "rewards/count_adherence/std": 0.027230001986026764, "rewards/meter/mean": 0.9936738014221191, "rewards/meter/std": 0.003524529282003641, "rewards/repeat_penalty/mean": 0.5536096096038818, "rewards/repeat_penalty/std": 0.012604749761521816, "rewards/total_composite/mean": 0.44484955072402954, "rewards/total_composite/std": 0.015336754731833935, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 994 }, { "clip_ratio/high_max": 0.010277777910232544, "clip_ratio/high_mean": 0.010277777910232544, "clip_ratio/low_mean": 0.03413326805457473, "clip_ratio/low_min": 0.03413326805457473, "clip_ratio/region_mean": 0.04441104596480727, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 76.25, "completions/mean_terminated_length": 76.25, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.28803054243326187, "epoch": 0.03996465437602924, "frac_reward_zero_std": 0.0, "grad_norm": 5.63449764251709, "learning_rate": 6.987878787878788e-06, "loss": 0.0319, "num_tokens": 2234891.0, "reward": 0.03228817135095596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.03228817135095596, "reward_meter_std": 0.06514540314674377, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06514539569616318, "reward_total_composite_mean": 0.03228817135095596, "reward_total_composite_std": 0.06514540314674377, "reward_total_mean": 0.03228817135095596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.03228817135095596, "rewards/meter/std": 0.06514540314674377, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.03228817135095596, "rewards/total_composite/std": 0.06514540314674377, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0099796056747437, "sampling/importance_sampling_ratio/min": 0.11422906816005707, "sampling/sampling_logp_difference/max": 2.1695494651794434, "sampling/sampling_logp_difference/mean": 0.0636926218867302, "step": 995 }, { "clip_ratio/high_max": 0.01975645322818309, "clip_ratio/high_mean": 0.01975645322818309, "clip_ratio/low_mean": 0.019090469810180366, "clip_ratio/low_min": 0.019090469810180366, "clip_ratio/region_mean": 0.03884692303836346, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 76.375, "completions/mean_terminated_length": 76.375, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.23532197065651417, "epoch": 0.04000481985781419, "frac_reward_zero_std": 0.0, "grad_norm": 6.409210205078125, "learning_rate": 6.984848484848485e-06, "loss": 0.0295, "num_tokens": 2236814.0, "reward": 0.010214247740805149, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.010214247740805149, "reward_meter_std": 0.007614050526171923, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007614050526171923, "reward_total_composite_mean": 0.010214247740805149, "reward_total_composite_std": 0.007614050526171923, "reward_total_mean": 0.010214247740805149, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.010214247740805149, "rewards/meter/std": 0.007614050526171923, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.010214247740805149, "rewards/total_composite/std": 0.007614050526171923, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0045915842056274, "sampling/importance_sampling_ratio/min": 0.21826720237731934, "sampling/sampling_logp_difference/max": 1.5220353603363037, "sampling/sampling_logp_difference/mean": 0.0485016293823719, "step": 996 }, { "clip_ratio/high_max": 0.009801336331292987, "clip_ratio/high_mean": 0.009801336331292987, "clip_ratio/low_mean": 0.03696049621794373, "clip_ratio/low_min": 0.03696049621794373, "clip_ratio/region_mean": 0.046761832549236715, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 77.625, "completions/mean_terminated_length": 77.625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.15790804475545883, "epoch": 0.040044985339599146, "frac_reward_zero_std": 0.0, "grad_norm": 5.538641929626465, "learning_rate": 6.981818181818183e-06, "loss": 0.0187, "num_tokens": 2238755.0, "reward": 0.002925167791545391, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.12371634691953659, "reward_meter_std": 0.3405037224292755, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.005130887031555176, "reward_total_composite_mean": 0.002925167791545391, "reward_total_composite_std": 0.005130887497216463, "reward_total_mean": 0.002925167791545391, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.12371634691953659, "rewards/meter/std": 0.3405037224292755, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.002925167791545391, "rewards/total_composite/std": 0.005130887497216463, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9927118420600891, "sampling/importance_sampling_ratio/min": 0.016660349443554878, "sampling/sampling_logp_difference/max": 4.094723701477051, "sampling/sampling_logp_difference/mean": 0.06538589298725128, "step": 997 }, { "clip_ratio/high_max": 0.020238095661625266, "clip_ratio/high_mean": 0.020238095661625266, "clip_ratio/low_mean": 0.009268068009987473, "clip_ratio/low_min": 0.009268068009987473, "clip_ratio/region_mean": 0.02950616367161274, "completions/clipped_ratio": 0.0, "completions/max_length": 45.0, "completions/max_terminated_length": 45.0, "completions/mean_length": 42.5, "completions/mean_terminated_length": 42.5, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.2211740417405963, "epoch": 0.0400851508213841, "frac_reward_zero_std": 0.0, "grad_norm": 7.545363903045654, "learning_rate": 6.978787878787879e-06, "loss": -0.0461, "num_tokens": 2240295.0, "reward": 0.9984815120697021, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984815120697021, "reward_meter_std": 0.001035381923429668, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010353842517361045, "reward_total_composite_mean": 0.9984815120697021, "reward_total_composite_std": 0.001035381923429668, "reward_total_mean": 0.9984815120697021, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984815120697021, "rewards/meter/std": 0.001035381923429668, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984815120697021, "rewards/total_composite/std": 0.001035381923429668, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9990119934082031, "sampling/importance_sampling_ratio/min": 0.19786088168621063, "sampling/sampling_logp_difference/max": 1.6201910972595215, "sampling/sampling_logp_difference/mean": 0.043179791420698166, "step": 998 }, { "clip_ratio/high_max": 0.026683697244152427, "clip_ratio/high_mean": 0.026683697244152427, "clip_ratio/low_mean": 0.003737623686902225, "clip_ratio/low_min": 0.003737623686902225, "clip_ratio/region_mean": 0.030421320931054652, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 99.125, "completions/mean_terminated_length": 99.125, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.22251357324421406, "epoch": 0.040125316303169054, "frac_reward_zero_std": 0.0, "grad_norm": 7.812422752380371, "learning_rate": 6.975757575757577e-06, "loss": 0.0238, "num_tokens": 2242408.0, "reward": 0.8570353984832764, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8570353984832764, "reward_meter_std": 0.18506589531898499, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.18506589531898499, "reward_total_composite_mean": 0.8570353984832764, "reward_total_composite_std": 0.18506589531898499, "reward_total_mean": 0.8570353984832764, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8570353984832764, "rewards/meter/std": 0.18506589531898499, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8570353984832764, "rewards/total_composite/std": 0.18506589531898499, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0009180307388306, "sampling/importance_sampling_ratio/min": 0.2936861515045166, "sampling/sampling_logp_difference/max": 1.2252435684204102, "sampling/sampling_logp_difference/mean": 0.039642542600631714, "step": 999 }, { "clip_ratio/high_max": 0.0014534883666783571, "clip_ratio/high_mean": 0.0014534883666783571, "clip_ratio/low_mean": 0.01114646252244711, "clip_ratio/low_min": 0.01114646252244711, "clip_ratio/region_mean": 0.012599950889125466, "completions/clipped_ratio": 0.0, "completions/max_length": 92.0, "completions/max_terminated_length": 92.0, "completions/mean_length": 89.125, "completions/mean_terminated_length": 89.125, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.0689164032228291, "epoch": 0.04016548178495401, "frac_reward_zero_std": 0.0, "grad_norm": 2.9554319381713867, "learning_rate": 6.9727272727272735e-06, "loss": 0.016, "num_tokens": 2244489.0, "reward": 0.8204019665718079, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9947108030319214, "reward_meter_std": 0.004218456335365772, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06665877997875214, "reward_total_composite_mean": 0.8204019665718079, "reward_total_composite_std": 0.06665877252817154, "reward_total_mean": 0.8204019665718079, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9947108030319214, "rewards/meter/std": 0.004218456335365772, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8204019665718079, "rewards/total_composite/std": 0.06665877252817154, "sampling/importance_sampling_ratio/max": 1.6871954202651978, "sampling/importance_sampling_ratio/mean": 1.0028539896011353, "sampling/importance_sampling_ratio/min": 0.27113983035087585, "sampling/sampling_logp_difference/max": 1.3051207065582275, "sampling/sampling_logp_difference/mean": 0.015147675760090351, "step": 1000 }, { "epoch": 0.04016548178495401, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.04807692307692308, "eval_completions/max_length": 449.38461538461536, "eval_completions/max_terminated_length": 413.2307692307692, "eval_completions/mean_length": 233.69230769230768, "eval_completions/mean_terminated_length": 219.62088364821213, "eval_completions/min_length": 60.0, "eval_completions/min_terminated_length": 60.0, "eval_entropy": 0.07349483840740643, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 2244489.0, "eval_reward": 0.42624477927501386, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.9180091665341303, "eval_reward_count_adherence_std": 0.09919005589416394, "eval_reward_meter_mean": 0.644585329752702, "eval_reward_meter_std": 0.44141573172349197, "eval_reward_repeat_penalty_mean": 0.7226548378284161, "eval_reward_repeat_penalty_std": 0.17840037838770792, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.42624477927501386, "eval_reward_total_composite_std": 0.3310052202298091, "eval_reward_total_mean": 0.42624477927501386, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.9180091665341303, "eval_rewards/count_adherence/std": 0.09919005589416394, "eval_rewards/meter/mean": 0.644585329752702, "eval_rewards/meter/std": 0.44141573172349197, "eval_rewards/repeat_penalty/mean": 0.7226548378284161, "eval_rewards/repeat_penalty/std": 0.17840037838770792, "eval_rewards/total_composite/mean": 0.42624477927501386, "eval_rewards/total_composite/std": 0.3310052202298091, "eval_runtime": 83.6634, "eval_samples_per_second": 1.243, "eval_sampling/importance_sampling_ratio/max": 1.3331910921977117, "eval_sampling/importance_sampling_ratio/mean": 1.0016976503225474, "eval_sampling/importance_sampling_ratio/min": 0.4674220451941857, "eval_sampling/sampling_logp_difference/max": 0.8023634048608633, "eval_sampling/sampling_logp_difference/mean": 0.00861521104637247, "eval_steps_per_second": 0.155, "step": 1000 }, { "clip_ratio/high_max": 0.02217741869390011, "clip_ratio/high_mean": 0.02217741869390011, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/region_mean": 0.02803679369390011, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.25, "completions/mean_terminated_length": 62.25, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.13137870281934738, "epoch": 0.04020564726673896, "frac_reward_zero_std": 0.0, "grad_norm": 8.061637878417969, "learning_rate": 6.969696969696971e-06, "loss": 0.0233, "num_tokens": 2246187.0, "reward": 0.8264527320861816, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8264527320861816, "reward_meter_std": 0.1885237991809845, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1885237991809845, "reward_total_composite_mean": 0.8264527320861816, "reward_total_composite_std": 0.1885237991809845, "reward_total_mean": 0.8264527320861816, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8264527320861816, "rewards/meter/std": 0.1885237991809845, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8264527320861816, "rewards/total_composite/std": 0.1885237991809845, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0087472200393677, "sampling/importance_sampling_ratio/min": 0.38382700085639954, "sampling/sampling_logp_difference/max": 1.3253092765808105, "sampling/sampling_logp_difference/mean": 0.02593499980866909, "step": 1001 }, { "clip_ratio/high_max": 0.03800246538594365, "clip_ratio/high_mean": 0.03800246538594365, "clip_ratio/low_mean": 0.012928195297718048, "clip_ratio/low_min": 0.012928195297718048, "clip_ratio/region_mean": 0.0509306606836617, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.23851481638848782, "epoch": 0.040245812748523915, "frac_reward_zero_std": 0.0, "grad_norm": 9.637553215026855, "learning_rate": 6.966666666666667e-06, "loss": 0.0292, "num_tokens": 2247834.0, "reward": 0.8884119987487793, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8884119987487793, "reward_meter_std": 0.23295524716377258, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2329552173614502, "reward_total_composite_mean": 0.8884119987487793, "reward_total_composite_std": 0.23295524716377258, "reward_total_mean": 0.8884119987487793, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8884119987487793, "rewards/meter/std": 0.23295524716377258, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8884119987487793, "rewards/total_composite/std": 0.23295524716377258, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028692483901978, "sampling/importance_sampling_ratio/min": 0.16747726500034332, "sampling/sampling_logp_difference/max": 1.78690767288208, "sampling/sampling_logp_difference/mean": 0.06217576563358307, "step": 1002 }, { "clip_ratio/high_max": 0.005464061861857772, "clip_ratio/high_mean": 0.005464061861857772, "clip_ratio/low_mean": 0.004366895416751504, "clip_ratio/low_min": 0.004366895416751504, "clip_ratio/region_mean": 0.009830957278609276, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 114.625, "completions/mean_terminated_length": 114.625, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.05840234411880374, "epoch": 0.04028597823030887, "frac_reward_zero_std": 0.0, "grad_norm": 5.540070056915283, "learning_rate": 6.963636363636364e-06, "loss": 0.0045, "num_tokens": 2250215.0, "reward": 0.634056806564331, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.781272828578949, "reward_meter_std": 0.39195698499679565, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.31964385509490967, "reward_total_composite_mean": 0.634056806564331, "reward_total_composite_std": 0.31964385509490967, "reward_total_mean": 0.634056806564331, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.781272828578949, "rewards/meter/std": 0.39195698499679565, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.634056806564331, "rewards/total_composite/std": 0.31964385509490967, "sampling/importance_sampling_ratio/max": 1.965267539024353, "sampling/importance_sampling_ratio/mean": 0.9998385906219482, "sampling/importance_sampling_ratio/min": 0.22292707860469818, "sampling/sampling_logp_difference/max": 1.5009106397628784, "sampling/sampling_logp_difference/mean": 0.014822704717516899, "step": 1003 }, { "clip_ratio/high_max": 0.014583333861082792, "clip_ratio/high_mean": 0.014583333861082792, "clip_ratio/low_mean": 0.004273816477507353, "clip_ratio/low_min": 0.004273816477507353, "clip_ratio/region_mean": 0.018857150338590145, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 59.625, "completions/mean_terminated_length": 59.625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.06448704563081264, "epoch": 0.04032614371209382, "frac_reward_zero_std": 0.0, "grad_norm": 7.136742115020752, "learning_rate": 6.960606060606061e-06, "loss": 0.0036, "num_tokens": 2251916.0, "reward": 0.9974349737167358, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974349737167358, "reward_meter_std": 0.0002844088012352586, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002844203554559499, "reward_total_composite_mean": 0.9974349737167358, "reward_total_composite_std": 0.0002844088012352586, "reward_total_mean": 0.9974349737167358, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974349737167358, "rewards/meter/std": 0.0002844088012352586, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974349737167358, "rewards/total_composite/std": 0.0002844088012352586, "sampling/importance_sampling_ratio/max": 1.5021758079528809, "sampling/importance_sampling_ratio/mean": 0.9971798062324524, "sampling/importance_sampling_ratio/min": 0.25123220682144165, "sampling/sampling_logp_difference/max": 1.3813776969909668, "sampling/sampling_logp_difference/mean": 0.018237588927149773, "step": 1004 }, { "clip_ratio/high_max": 0.014221564109902829, "clip_ratio/high_mean": 0.014221564109902829, "clip_ratio/low_mean": 0.005632697488181293, "clip_ratio/low_min": 0.005632697488181293, "clip_ratio/region_mean": 0.019854261598084122, "completions/clipped_ratio": 0.0, "completions/max_length": 203.0, "completions/max_terminated_length": 203.0, "completions/mean_length": 197.125, "completions/mean_terminated_length": 197.125, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.11270508635789156, "epoch": 0.04036630919387878, "frac_reward_zero_std": 0.0, "grad_norm": 2.455113649368286, "learning_rate": 6.957575757575759e-06, "loss": 0.0226, "num_tokens": 2255101.0, "reward": 0.6168004274368286, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9832211136817932, "reward_meter_std": 0.004388164728879929, "reward_repeat_penalty_mean": 0.7840908765792847, "reward_repeat_penalty_std": 0.06763852387666702, "reward_std": 0.054140083491802216, "reward_total_composite_mean": 0.6168004274368286, "reward_total_composite_std": 0.054140087217092514, "reward_total_mean": 0.6168004274368286, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9832211136817932, "rewards/meter/std": 0.004388164728879929, "rewards/repeat_penalty/mean": 0.7840908765792847, "rewards/repeat_penalty/std": 0.06763852387666702, "rewards/total_composite/mean": 0.6168004274368286, "rewards/total_composite/std": 0.054140087217092514, "sampling/importance_sampling_ratio/max": 1.8359591960906982, "sampling/importance_sampling_ratio/mean": 1.0025383234024048, "sampling/importance_sampling_ratio/min": 0.27628687024116516, "sampling/sampling_logp_difference/max": 1.2863155603408813, "sampling/sampling_logp_difference/mean": 0.023318301886320114, "step": 1005 }, { "clip_ratio/high_max": 0.010740866768173873, "clip_ratio/high_mean": 0.010740866768173873, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/region_mean": 0.013865866814740002, "completions/clipped_ratio": 0.0, "completions/max_length": 95.0, "completions/max_terminated_length": 95.0, "completions/mean_length": 91.375, "completions/mean_terminated_length": 91.375, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.09509005583822727, "epoch": 0.04040647467566374, "frac_reward_zero_std": 0.0, "grad_norm": 7.012386798858643, "learning_rate": 6.954545454545455e-06, "loss": -0.0394, "num_tokens": 2257080.0, "reward": 0.7331008911132812, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9163761734962463, "reward_meter_std": 0.1686118096113205, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1348894238471985, "reward_total_composite_mean": 0.7331008911132812, "reward_total_composite_std": 0.13488943874835968, "reward_total_mean": 0.7331008911132812, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9163761734962463, "rewards/meter/std": 0.1686118096113205, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7331008911132812, "rewards/total_composite/std": 0.13488943874835968, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9956486225128174, "sampling/importance_sampling_ratio/min": 0.16208584606647491, "sampling/sampling_logp_difference/max": 1.819629192352295, "sampling/sampling_logp_difference/mean": 0.024793440476059914, "step": 1006 }, { "clip_ratio/high_max": 0.008266129298135638, "clip_ratio/high_mean": 0.008266129298135638, "clip_ratio/low_mean": 0.004166666883975267, "clip_ratio/low_min": 0.004166666883975267, "clip_ratio/region_mean": 0.012432796182110906, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 60.25, "completions/mean_terminated_length": 60.25, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.04919788311235607, "epoch": 0.04044664015744869, "frac_reward_zero_std": 0.0, "grad_norm": 2.0443665981292725, "learning_rate": 6.951515151515153e-06, "loss": -0.0001, "num_tokens": 2258850.0, "reward": 0.9974480867385864, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974480867385864, "reward_meter_std": 0.00012538061127997935, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00012537713337223977, "reward_total_composite_mean": 0.9974480867385864, "reward_total_composite_std": 0.00012538061127997935, "reward_total_mean": 0.9974480867385864, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974480867385864, "rewards/meter/std": 0.00012538061127997935, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974480867385864, "rewards/total_composite/std": 0.00012538061127997935, "sampling/importance_sampling_ratio/max": 1.6381621360778809, "sampling/importance_sampling_ratio/mean": 0.9988527894020081, "sampling/importance_sampling_ratio/min": 0.4883781373500824, "sampling/sampling_logp_difference/max": 0.7166652679443359, "sampling/sampling_logp_difference/mean": 0.013044199906289577, "step": 1007 }, { "clip_ratio/high_max": 0.003859857562929392, "clip_ratio/high_mean": 0.003859857562929392, "clip_ratio/low_mean": 0.010143639519810677, "clip_ratio/low_min": 0.010143639519810677, "clip_ratio/region_mean": 0.014003497082740068, "completions/clipped_ratio": 0.0, "completions/max_length": 471.0, "completions/max_terminated_length": 471.0, "completions/mean_length": 423.0, "completions/mean_terminated_length": 423.0, "completions/min_length": 336.0, "completions/min_terminated_length": 336.0, "entropy": 0.08116073720157146, "epoch": 0.040486805639233646, "frac_reward_zero_std": 0.0, "grad_norm": 2.1715681552886963, "learning_rate": 6.948484848484849e-06, "loss": 0.0083, "num_tokens": 2263802.0, "reward": 0.017225507646799088, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9107142686843872, "reward_count_adherence_std": 0.0832117572426796, "reward_meter_mean": 0.029939208179712296, "reward_meter_std": 0.05622902140021324, "reward_repeat_penalty_mean": 0.5447744727134705, "reward_repeat_penalty_std": 0.19969096779823303, "reward_std": 0.036213167011737823, "reward_total_composite_mean": 0.017225507646799088, "reward_total_composite_std": 0.03621317073702812, "reward_total_mean": 0.017225507646799088, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9107142686843872, "rewards/count_adherence/std": 0.0832117572426796, "rewards/meter/mean": 0.029939208179712296, "rewards/meter/std": 0.05622902140021324, "rewards/repeat_penalty/mean": 0.5447744727134705, "rewards/repeat_penalty/std": 0.19969096779823303, "rewards/total_composite/mean": 0.017225507646799088, "rewards/total_composite/std": 0.03621317073702812, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0038172006607056, "sampling/importance_sampling_ratio/min": 0.061861179769039154, "sampling/sampling_logp_difference/max": 2.782862424850464, "sampling/sampling_logp_difference/mean": 0.014329123310744762, "step": 1008 }, { "clip_ratio/high_max": 0.004261363763362169, "clip_ratio/high_mean": 0.004261363763362169, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/region_mean": 0.006978755118325353, "completions/clipped_ratio": 0.0, "completions/max_length": 92.0, "completions/max_terminated_length": 92.0, "completions/mean_length": 88.75, "completions/mean_terminated_length": 88.75, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.02403299231082201, "epoch": 0.0405269711210186, "frac_reward_zero_std": 0.0, "grad_norm": 6.43966007232666, "learning_rate": 6.945454545454546e-06, "loss": 0.0182, "num_tokens": 2265904.0, "reward": 0.7726212739944458, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9968655705451965, "reward_meter_std": 0.0008970692288130522, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07099878042936325, "reward_total_composite_mean": 0.7726212739944458, "reward_total_composite_std": 0.07099877297878265, "reward_total_mean": 0.7726212739944458, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9968655705451965, "rewards/meter/std": 0.0008970692288130522, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7726212739944458, "rewards/total_composite/std": 0.07099877297878265, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016840696334839, "sampling/importance_sampling_ratio/min": 0.6927711367607117, "sampling/sampling_logp_difference/max": 0.8346314430236816, "sampling/sampling_logp_difference/mean": 0.006405611056834459, "step": 1009 }, { "clip_ratio/high_max": 0.023642246145755053, "clip_ratio/high_mean": 0.023642246145755053, "clip_ratio/low_mean": 0.01056338008493185, "clip_ratio/low_min": 0.01056338008493185, "clip_ratio/region_mean": 0.0342056262306869, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.5, "completions/mean_terminated_length": 69.5, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.22077207453548908, "epoch": 0.040567136602803554, "frac_reward_zero_std": 0.0, "grad_norm": 7.395992755889893, "learning_rate": 6.942424242424243e-06, "loss": 0.0216, "num_tokens": 2267676.0, "reward": 0.9021741151809692, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9437720775604248, "reward_meter_std": 0.10555455833673477, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.14069606363773346, "reward_total_composite_mean": 0.9021741151809692, "reward_total_composite_std": 0.14069606363773346, "reward_total_mean": 0.9021741151809692, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9437720775604248, "rewards/meter/std": 0.10555455833673477, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9021741151809692, "rewards/total_composite/std": 0.14069606363773346, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0084481239318848, "sampling/importance_sampling_ratio/min": 0.2674984335899353, "sampling/sampling_logp_difference/max": 1.3186415433883667, "sampling/sampling_logp_difference/mean": 0.0496804304420948, "step": 1010 }, { "clip_ratio/high_max": 0.047181503381580114, "clip_ratio/high_mean": 0.047181503381580114, "clip_ratio/low_mean": 0.018277311231940985, "clip_ratio/low_min": 0.018277311231940985, "clip_ratio/region_mean": 0.0654588146135211, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 68.5, "completions/mean_terminated_length": 68.5, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2325905654579401, "epoch": 0.04060730208458851, "frac_reward_zero_std": 0.0, "grad_norm": 5.829395771026611, "learning_rate": 6.93939393939394e-06, "loss": 0.0142, "num_tokens": 2269520.0, "reward": 0.9850254058837891, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9850254058837891, "reward_meter_std": 0.01548534631729126, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.015485321171581745, "reward_total_composite_mean": 0.9850254058837891, "reward_total_composite_std": 0.01548534631729126, "reward_total_mean": 0.9850254058837891, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9850254058837891, "rewards/meter/std": 0.01548534631729126, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9850254058837891, "rewards/total_composite/std": 0.01548534631729126, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9924511313438416, "sampling/importance_sampling_ratio/min": 0.1980629861354828, "sampling/sampling_logp_difference/max": 1.6191701889038086, "sampling/sampling_logp_difference/mean": 0.06408417969942093, "step": 1011 }, { "clip_ratio/high_max": 0.03817800944671035, "clip_ratio/high_mean": 0.03817800944671035, "clip_ratio/low_mean": 0.01664550113491714, "clip_ratio/low_min": 0.01664550113491714, "clip_ratio/region_mean": 0.05482351058162749, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 102.75, "completions/mean_terminated_length": 102.75, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.30692350305616856, "epoch": 0.04064746756637346, "frac_reward_zero_std": 0.0, "grad_norm": 5.250216960906982, "learning_rate": 6.936363636363636e-06, "loss": 0.0315, "num_tokens": 2271718.0, "reward": 0.9896959066390991, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9896959066390991, "reward_meter_std": 0.009748869575560093, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009748890995979309, "reward_total_composite_mean": 0.9896959066390991, "reward_total_composite_std": 0.009748869575560093, "reward_total_mean": 0.9896959066390991, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9896959066390991, "rewards/meter/std": 0.009748869575560093, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9896959066390991, "rewards/total_composite/std": 0.009748869575560093, "sampling/importance_sampling_ratio/max": 1.915394902229309, "sampling/importance_sampling_ratio/mean": 1.0056629180908203, "sampling/importance_sampling_ratio/min": 0.33834466338157654, "sampling/sampling_logp_difference/max": 1.0836901664733887, "sampling/sampling_logp_difference/mean": 0.05327339842915535, "step": 1012 }, { "clip_ratio/high_max": 0.00749256182461977, "clip_ratio/high_mean": 0.00749256182461977, "clip_ratio/low_mean": 0.01979458425194025, "clip_ratio/low_min": 0.01979458425194025, "clip_ratio/region_mean": 0.02728714607656002, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 114.0, "completions/mean_terminated_length": 114.0, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.25786815397441387, "epoch": 0.040687633048158416, "frac_reward_zero_std": 0.0, "grad_norm": 5.079646110534668, "learning_rate": 6.9333333333333344e-06, "loss": -0.0071, "num_tokens": 2273974.0, "reward": 0.0039320518262684345, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.004421628080308437, "reward_meter_std": 0.004275813698768616, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.0040371594950556755, "reward_total_composite_mean": 0.0039320518262684345, "reward_total_composite_std": 0.004037159029394388, "reward_total_mean": 0.0039320518262684345, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.004421628080308437, "rewards/meter/std": 0.004275813698768616, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.0039320518262684345, "rewards/total_composite/std": 0.004037159029394388, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0108870267868042, "sampling/importance_sampling_ratio/min": 0.2919834554195404, "sampling/sampling_logp_difference/max": 1.231058120727539, "sampling/sampling_logp_difference/mean": 0.035947732627391815, "step": 1013 }, { "clip_ratio/high_max": 0.020599488052539527, "clip_ratio/high_mean": 0.020599488052539527, "clip_ratio/low_mean": 0.005090707214549184, "clip_ratio/low_min": 0.005090707214549184, "clip_ratio/region_mean": 0.02569019526708871, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 74.375, "completions/mean_terminated_length": 74.375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.16714145429432392, "epoch": 0.04072779852994337, "frac_reward_zero_std": 0.0, "grad_norm": 3.9192073345184326, "learning_rate": 6.930303030303031e-06, "loss": 0.006, "num_tokens": 2275905.0, "reward": 0.9924027919769287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9924027919769287, "reward_meter_std": 0.005311480723321438, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005311472807079554, "reward_total_composite_mean": 0.9924027919769287, "reward_total_composite_std": 0.005311480723321438, "reward_total_mean": 0.9924027919769287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9924027919769287, "rewards/meter/std": 0.005311480723321438, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924027919769287, "rewards/total_composite/std": 0.005311480723321438, "sampling/importance_sampling_ratio/max": 1.5236577987670898, "sampling/importance_sampling_ratio/mean": 1.0100598335266113, "sampling/importance_sampling_ratio/min": 0.5449473261833191, "sampling/sampling_logp_difference/max": 0.6070661544799805, "sampling/sampling_logp_difference/mean": 0.024386199191212654, "step": 1014 }, { "clip_ratio/high_max": 0.007247899193316698, "clip_ratio/high_mean": 0.007247899193316698, "clip_ratio/low_mean": 0.007352941203862429, "clip_ratio/low_min": 0.007352941203862429, "clip_ratio/region_mean": 0.014600840397179127, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 33.75, "completions/mean_terminated_length": 33.75, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.22182095050811768, "epoch": 0.04076796401172832, "frac_reward_zero_std": 0.0, "grad_norm": 6.082520008087158, "learning_rate": 6.927272727272728e-06, "loss": 0.0154, "num_tokens": 2277383.0, "reward": 0.9623132944107056, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9623132944107056, "reward_meter_std": 0.02709752693772316, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.027097532525658607, "reward_total_composite_mean": 0.9623132944107056, "reward_total_composite_std": 0.02709752693772316, "reward_total_mean": 0.9623132944107056, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9623132944107056, "rewards/meter/std": 0.02709752693772316, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9623132944107056, "rewards/total_composite/std": 0.02709752693772316, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01416015625, "sampling/importance_sampling_ratio/min": 0.3405103385448456, "sampling/sampling_logp_difference/max": 1.1891875267028809, "sampling/sampling_logp_difference/mean": 0.03504392132163048, "step": 1015 }, { "clip_ratio/high_max": 0.00996917171869427, "clip_ratio/high_mean": 0.00996917171869427, "clip_ratio/low_mean": 0.002274398400913924, "clip_ratio/low_min": 0.002274398400913924, "clip_ratio/region_mean": 0.012243570119608194, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 400.0, "completions/mean_length": 404.125, "completions/mean_terminated_length": 388.71429443359375, "completions/min_length": 374.0, "completions/min_terminated_length": 374.0, "entropy": 0.04754339391365647, "epoch": 0.04080812949351328, "frac_reward_zero_std": 0.0, "grad_norm": 1.2893778085708618, "learning_rate": 6.9242424242424245e-06, "loss": -0.1996, "num_tokens": 2281936.0, "reward": 0.37017643451690674, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8083332777023315, "reward_count_adherence_std": 0.27472931146621704, "reward_meter_mean": 0.8926552534103394, "reward_meter_std": 0.2755105197429657, "reward_repeat_penalty_mean": 0.5423148274421692, "reward_repeat_penalty_std": 0.26660624146461487, "reward_std": 0.22237010300159454, "reward_total_composite_mean": 0.37017643451690674, "reward_total_composite_std": 0.22237010300159454, "reward_total_mean": 0.37017643451690674, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8083332777023315, "rewards/count_adherence/std": 0.27472931146621704, "rewards/meter/mean": 0.8926552534103394, "rewards/meter/std": 0.2755105197429657, "rewards/repeat_penalty/mean": 0.5423148274421692, "rewards/repeat_penalty/std": 0.26660624146461487, "rewards/total_composite/mean": 0.37017643451690674, "rewards/total_composite/std": 0.22237010300159454, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002998948097229, "sampling/importance_sampling_ratio/min": 0.035339467227458954, "sampling/sampling_logp_difference/max": 3.34275484085083, "sampling/sampling_logp_difference/mean": 0.015432414598762989, "step": 1016 }, { "clip_ratio/high_max": 0.02297498215921223, "clip_ratio/high_mean": 0.02297498215921223, "clip_ratio/low_mean": 0.013492224738001823, "clip_ratio/low_min": 0.013492224738001823, "clip_ratio/region_mean": 0.036467206897214055, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 75.0, "completions/mean_terminated_length": 75.0, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.2947879508137703, "epoch": 0.04084829497529823, "frac_reward_zero_std": 0.0, "grad_norm": 4.289333343505859, "learning_rate": 6.921212121212122e-06, "loss": -0.0018, "num_tokens": 2283872.0, "reward": 0.9958159327507019, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958159327507019, "reward_meter_std": 0.0007772990502417088, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007773060933686793, "reward_total_composite_mean": 0.9958159327507019, "reward_total_composite_std": 0.0007772990502417088, "reward_total_mean": 0.9958159327507019, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958159327507019, "rewards/meter/std": 0.0007772990502417088, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958159327507019, "rewards/total_composite/std": 0.0007772990502417088, "sampling/importance_sampling_ratio/max": 1.9063295125961304, "sampling/importance_sampling_ratio/mean": 1.0068724155426025, "sampling/importance_sampling_ratio/min": 0.26935896277427673, "sampling/sampling_logp_difference/max": 1.3117103576660156, "sampling/sampling_logp_difference/mean": 0.05131428688764572, "step": 1017 }, { "clip_ratio/high_max": 0.057335917837917805, "clip_ratio/high_mean": 0.057335917837917805, "clip_ratio/low_mean": 0.02322303969413042, "clip_ratio/low_min": 0.02322303969413042, "clip_ratio/region_mean": 0.08055895753204823, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 69.625, "completions/mean_terminated_length": 69.625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.4668789766728878, "epoch": 0.040888460457083185, "frac_reward_zero_std": 0.0, "grad_norm": 6.446331977844238, "learning_rate": 6.918181818181818e-06, "loss": 0.0122, "num_tokens": 2285685.0, "reward": 0.9972782135009766, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972782135009766, "reward_meter_std": 0.0013893006835132837, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0013893075520172715, "reward_total_composite_mean": 0.9972782135009766, "reward_total_composite_std": 0.0013893006835132837, "reward_total_mean": 0.9972782135009766, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972782135009766, "rewards/meter/std": 0.0013893006835132837, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972782135009766, "rewards/total_composite/std": 0.0013893006835132837, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028208494186401, "sampling/importance_sampling_ratio/min": 0.0022041599731892347, "sampling/sampling_logp_difference/max": 6.117408752441406, "sampling/sampling_logp_difference/mean": 0.09435711801052094, "step": 1018 }, { "clip_ratio/high_max": 0.023812503553926945, "clip_ratio/high_mean": 0.023812503553926945, "clip_ratio/low_mean": 0.02015177276916802, "clip_ratio/low_min": 0.02015177276916802, "clip_ratio/region_mean": 0.043964276323094964, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 82.875, "completions/mean_terminated_length": 82.875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.2028534710407257, "epoch": 0.04092862593886814, "frac_reward_zero_std": 0.0, "grad_norm": 8.041248321533203, "learning_rate": 6.915151515151515e-06, "loss": -0.0074, "num_tokens": 2287596.0, "reward": 0.42031192779541016, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.42031192779541016, "reward_meter_std": 0.37385284900665283, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.37385284900665283, "reward_total_composite_mean": 0.42031192779541016, "reward_total_composite_std": 0.37385284900665283, "reward_total_mean": 0.42031192779541016, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.42031192779541016, "rewards/meter/std": 0.37385284900665283, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.42031192779541016, "rewards/total_composite/std": 0.37385284900665283, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9995813965797424, "sampling/importance_sampling_ratio/min": 0.221580371260643, "sampling/sampling_logp_difference/max": 1.506969928741455, "sampling/sampling_logp_difference/mean": 0.04477884620428085, "step": 1019 }, { "clip_ratio/high_max": 0.01801690785214305, "clip_ratio/high_mean": 0.01801690785214305, "clip_ratio/low_mean": 0.0065684852888807654, "clip_ratio/low_min": 0.0065684852888807654, "clip_ratio/region_mean": 0.024585393141023815, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 115.75, "completions/mean_terminated_length": 115.75, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.20292328391224146, "epoch": 0.04096879142065309, "frac_reward_zero_std": 0.0, "grad_norm": 3.4062325954437256, "learning_rate": 6.912121212121212e-06, "loss": 0.0039, "num_tokens": 2290034.0, "reward": 0.8668707013130188, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9905360341072083, "reward_meter_std": 0.011200688779354095, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10446702688932419, "reward_total_composite_mean": 0.8668707013130188, "reward_total_composite_std": 0.10446701943874359, "reward_total_mean": 0.8668707013130188, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9905360341072083, "rewards/meter/std": 0.011200688779354095, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8668707013130188, "rewards/total_composite/std": 0.10446701943874359, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0034315586090088, "sampling/importance_sampling_ratio/min": 0.3525734841823578, "sampling/sampling_logp_difference/max": 1.1195735931396484, "sampling/sampling_logp_difference/mean": 0.031655244529247284, "step": 1020 }, { "clip_ratio/high_max": 0.023018090520054102, "clip_ratio/high_mean": 0.023018090520054102, "clip_ratio/low_mean": 0.022817460587248206, "clip_ratio/low_min": 0.022817460587248206, "clip_ratio/region_mean": 0.04583555110730231, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2529278863221407, "epoch": 0.04100895690243805, "frac_reward_zero_std": 0.0, "grad_norm": 5.57641077041626, "learning_rate": 6.90909090909091e-06, "loss": -0.0207, "num_tokens": 2291779.0, "reward": 0.1570926308631897, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.1570926308631897, "reward_meter_std": 0.11832874268293381, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1183287501335144, "reward_total_composite_mean": 0.1570926308631897, "reward_total_composite_std": 0.11832874268293381, "reward_total_mean": 0.1570926308631897, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.1570926308631897, "rewards/meter/std": 0.11832874268293381, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.1570926308631897, "rewards/total_composite/std": 0.11832874268293381, "sampling/importance_sampling_ratio/max": 1.6118385791778564, "sampling/importance_sampling_ratio/mean": 1.0030486583709717, "sampling/importance_sampling_ratio/min": 0.44465112686157227, "sampling/sampling_logp_difference/max": 0.8104653358459473, "sampling/sampling_logp_difference/mean": 0.038187529891729355, "step": 1021 }, { "clip_ratio/high_max": 0.02277943748049438, "clip_ratio/high_mean": 0.02277943748049438, "clip_ratio/low_mean": 0.008588795084506273, "clip_ratio/low_min": 0.008588795084506273, "clip_ratio/region_mean": 0.03136823256500065, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 82.375, "completions/mean_terminated_length": 82.375, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.21936733089387417, "epoch": 0.041049122384223, "frac_reward_zero_std": 0.0, "grad_norm": 3.1528494358062744, "learning_rate": 6.906060606060606e-06, "loss": 0.0374, "num_tokens": 2293742.0, "reward": 0.9961868524551392, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961868524551392, "reward_meter_std": 0.0018421995919197798, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0018422063440084457, "reward_total_composite_mean": 0.9961868524551392, "reward_total_composite_std": 0.0018421995919197798, "reward_total_mean": 0.9961868524551392, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961868524551392, "rewards/meter/std": 0.0018421995919197798, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961868524551392, "rewards/total_composite/std": 0.0018421995919197798, "sampling/importance_sampling_ratio/max": 1.9688695669174194, "sampling/importance_sampling_ratio/mean": 1.0033493041992188, "sampling/importance_sampling_ratio/min": 0.34118568897247314, "sampling/sampling_logp_difference/max": 1.0753283500671387, "sampling/sampling_logp_difference/mean": 0.036886442452669144, "step": 1022 }, { "clip_ratio/high_max": 0.023702416568994522, "clip_ratio/high_mean": 0.023702416568994522, "clip_ratio/low_mean": 0.019885645247995853, "clip_ratio/low_min": 0.019885645247995853, "clip_ratio/region_mean": 0.043588061816990376, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 68.75, "completions/mean_terminated_length": 68.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2538141794502735, "epoch": 0.041089287866007955, "frac_reward_zero_std": 0.0, "grad_norm": 5.055918216705322, "learning_rate": 6.903030303030304e-06, "loss": 0.0171, "num_tokens": 2295652.0, "reward": 0.2434961199760437, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2434961199760437, "reward_meter_std": 0.1991458386182785, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1991458386182785, "reward_total_composite_mean": 0.2434961199760437, "reward_total_composite_std": 0.1991458386182785, "reward_total_mean": 0.2434961199760437, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2434961199760437, "rewards/meter/std": 0.1991458386182785, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2434961199760437, "rewards/total_composite/std": 0.1991458386182785, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0007877349853516, "sampling/importance_sampling_ratio/min": 0.23404262959957123, "sampling/sampling_logp_difference/max": 1.4522520303726196, "sampling/sampling_logp_difference/mean": 0.05065210163593292, "step": 1023 }, { "clip_ratio/high_max": 0.006947804824449122, "clip_ratio/high_mean": 0.006947804824449122, "clip_ratio/low_mean": 0.004947002336848527, "clip_ratio/low_min": 0.004947002336848527, "clip_ratio/region_mean": 0.01189480716129765, "completions/clipped_ratio": 0.0, "completions/max_length": 406.0, "completions/max_terminated_length": 406.0, "completions/mean_length": 388.375, "completions/mean_terminated_length": 388.375, "completions/min_length": 351.0, "completions/min_terminated_length": 351.0, "entropy": 0.08917687484063208, "epoch": 0.04112945334779291, "frac_reward_zero_std": 0.0, "grad_norm": 0.9608837962150574, "learning_rate": 6.9e-06, "loss": -0.0145, "num_tokens": 2300311.0, "reward": 0.44603413343429565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.988775372505188, "reward_meter_std": 0.01991611160337925, "reward_repeat_penalty_mean": 0.6315789222717285, "reward_repeat_penalty_std": 0.05626552551984787, "reward_std": 0.0405924953520298, "reward_total_composite_mean": 0.44603413343429565, "reward_total_composite_std": 0.0405924916267395, "reward_total_mean": 0.44603413343429565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.988775372505188, "rewards/meter/std": 0.01991611160337925, "rewards/repeat_penalty/mean": 0.6315789222717285, "rewards/repeat_penalty/std": 0.05626552551984787, "rewards/total_composite/mean": 0.44603413343429565, "rewards/total_composite/std": 0.0405924916267395, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023077726364136, "sampling/importance_sampling_ratio/min": 0.22164598107337952, "sampling/sampling_logp_difference/max": 1.506673812866211, "sampling/sampling_logp_difference/mean": 0.012534739449620247, "step": 1024 }, { "clip_ratio/high_max": 0.011576289543882012, "clip_ratio/high_mean": 0.011576289543882012, "clip_ratio/low_mean": 0.01788918604142964, "clip_ratio/low_min": 0.01788918604142964, "clip_ratio/region_mean": 0.02946547558531165, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 97.625, "completions/mean_terminated_length": 97.625, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.2695270273834467, "epoch": 0.04116961882957786, "frac_reward_zero_std": 0.0, "grad_norm": 4.088324069976807, "learning_rate": 6.896969696969697e-06, "loss": 0.0116, "num_tokens": 2302372.0, "reward": 0.4717435836791992, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6049685478210449, "reward_meter_std": 0.33388465642929077, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.1511857807636261, "reward_std": 0.2634972631931305, "reward_total_composite_mean": 0.4717435836791992, "reward_total_composite_std": 0.2634972631931305, "reward_total_mean": 0.4717435836791992, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6049685478210449, "rewards/meter/std": 0.33388465642929077, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.1511857807636261, "rewards/total_composite/mean": 0.4717435836791992, "rewards/total_composite/std": 0.2634972631931305, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010300636291504, "sampling/importance_sampling_ratio/min": 0.23652127385139465, "sampling/sampling_logp_difference/max": 1.4417171478271484, "sampling/sampling_logp_difference/mean": 0.03984498605132103, "step": 1025 }, { "clip_ratio/high_max": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/low_mean": 0.020427904091775417, "clip_ratio/low_min": 0.020427904091775417, "clip_ratio/region_mean": 0.022381029091775417, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 61.25, "completions/mean_terminated_length": 61.25, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.15393794886767864, "epoch": 0.04120978431136282, "frac_reward_zero_std": 0.0, "grad_norm": 4.054835796356201, "learning_rate": 6.893939393939395e-06, "loss": -0.0083, "num_tokens": 2304286.0, "reward": 0.990164577960968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.990164577960968, "reward_meter_std": 0.004598760511726141, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004598761908710003, "reward_total_composite_mean": 0.990164577960968, "reward_total_composite_std": 0.004598760511726141, "reward_total_mean": 0.990164577960968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.990164577960968, "rewards/meter/std": 0.004598760511726141, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.990164577960968, "rewards/total_composite/std": 0.004598760511726141, "sampling/importance_sampling_ratio/max": 1.7953885793685913, "sampling/importance_sampling_ratio/mean": 1.0041474103927612, "sampling/importance_sampling_ratio/min": 0.3620084226131439, "sampling/sampling_logp_difference/max": 1.0160877704620361, "sampling/sampling_logp_difference/mean": 0.02800033800303936, "step": 1026 }, { "clip_ratio/high_max": 0.0187515887664631, "clip_ratio/high_mean": 0.0187515887664631, "clip_ratio/low_mean": 0.012531211483292282, "clip_ratio/low_min": 0.012531211483292282, "clip_ratio/region_mean": 0.03128280024975538, "completions/clipped_ratio": 0.0, "completions/max_length": 95.0, "completions/max_terminated_length": 95.0, "completions/mean_length": 90.625, "completions/mean_terminated_length": 90.625, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.18594534322619438, "epoch": 0.04124994979314777, "frac_reward_zero_std": 0.0, "grad_norm": 3.0363378524780273, "learning_rate": 6.890909090909092e-06, "loss": -0.0138, "num_tokens": 2306411.0, "reward": 0.8707464337348938, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952998757362366, "reward_meter_std": 0.0028919558972120285, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10161302983760834, "reward_total_composite_mean": 0.8707464337348938, "reward_total_composite_std": 0.10161302238702774, "reward_total_mean": 0.8707464337348938, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952998757362366, "rewards/meter/std": 0.0028919558972120285, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8707464337348938, "rewards/total_composite/std": 0.10161302238702774, "sampling/importance_sampling_ratio/max": 1.4683204889297485, "sampling/importance_sampling_ratio/mean": 0.9983088374137878, "sampling/importance_sampling_ratio/min": 0.23719395697116852, "sampling/sampling_logp_difference/max": 1.4388771057128906, "sampling/sampling_logp_difference/mean": 0.030290186405181885, "step": 1027 }, { "clip_ratio/high_max": 0.06568031571805477, "clip_ratio/high_mean": 0.06568031571805477, "clip_ratio/low_mean": 0.011456389911472797, "clip_ratio/low_min": 0.011456389911472797, "clip_ratio/region_mean": 0.07713670562952757, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.39760861173272133, "epoch": 0.041290115274932725, "frac_reward_zero_std": 0.0, "grad_norm": 12.675787925720215, "learning_rate": 6.887878787878789e-06, "loss": 0.0248, "num_tokens": 2308182.0, "reward": 0.9956432580947876, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956432580947876, "reward_meter_std": 0.003459326224401593, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003459327155724168, "reward_total_composite_mean": 0.9956432580947876, "reward_total_composite_std": 0.003459326224401593, "reward_total_mean": 0.9956432580947876, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956432580947876, "rewards/meter/std": 0.003459326224401593, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956432580947876, "rewards/total_composite/std": 0.003459326224401593, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0011348724365234, "sampling/importance_sampling_ratio/min": 0.12040401995182037, "sampling/sampling_logp_difference/max": 2.1169023513793945, "sampling/sampling_logp_difference/mean": 0.08104093372821808, "step": 1028 }, { "clip_ratio/high_max": 0.007978723384439945, "clip_ratio/high_mean": 0.007978723384439945, "clip_ratio/low_mean": 0.026484928792342544, "clip_ratio/low_min": 0.026484928792342544, "clip_ratio/region_mean": 0.03446365217678249, "completions/clipped_ratio": 0.0, "completions/max_length": 48.0, "completions/max_terminated_length": 48.0, "completions/mean_length": 47.0, "completions/mean_terminated_length": 47.0, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.34705137088894844, "epoch": 0.04133028075671768, "frac_reward_zero_std": 0.0, "grad_norm": 5.34241247177124, "learning_rate": 6.8848484848484854e-06, "loss": 0.0126, "num_tokens": 2309806.0, "reward": 0.12917472422122955, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.12917472422122955, "reward_meter_std": 0.3293028473854065, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3293028175830841, "reward_total_composite_mean": 0.12917472422122955, "reward_total_composite_std": 0.3293028473854065, "reward_total_mean": 0.12917472422122955, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.12917472422122955, "rewards/meter/std": 0.3293028473854065, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.12917472422122955, "rewards/total_composite/std": 0.3293028473854065, "sampling/importance_sampling_ratio/max": 1.494503378868103, "sampling/importance_sampling_ratio/mean": 0.9958956837654114, "sampling/importance_sampling_ratio/min": 0.11059728264808655, "sampling/sampling_logp_difference/max": 2.201859712600708, "sampling/sampling_logp_difference/mean": 0.07030355930328369, "step": 1029 }, { "clip_ratio/high_max": 0.032955562230199575, "clip_ratio/high_mean": 0.032955562230199575, "clip_ratio/low_mean": 0.02179320773575455, "clip_ratio/low_min": 0.02179320773575455, "clip_ratio/region_mean": 0.054748769965954125, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.125, "completions/mean_terminated_length": 70.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.4643949382007122, "epoch": 0.04137044623850263, "frac_reward_zero_std": 0.0, "grad_norm": 8.219462394714355, "learning_rate": 6.881818181818183e-06, "loss": -0.0076, "num_tokens": 2311551.0, "reward": 0.12689465284347534, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.12689465284347534, "reward_meter_std": 0.10111062228679657, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.10111062228679657, "reward_total_composite_mean": 0.12689465284347534, "reward_total_composite_std": 0.10111062228679657, "reward_total_mean": 0.12689465284347534, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.12689465284347534, "rewards/meter/std": 0.10111062228679657, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.12689465284347534, "rewards/total_composite/std": 0.10111062228679657, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0122308731079102, "sampling/importance_sampling_ratio/min": 0.19809359312057495, "sampling/sampling_logp_difference/max": 1.6190156936645508, "sampling/sampling_logp_difference/mean": 0.07496663182973862, "step": 1030 }, { "clip_ratio/high_max": 0.027175352559424937, "clip_ratio/high_mean": 0.027175352559424937, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.027175352559424937, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 224.875, "completions/mean_terminated_length": 129.1666717529297, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.16609951201826334, "epoch": 0.041410611720287586, "frac_reward_zero_std": 0.0, "grad_norm": 1.1240724325180054, "learning_rate": 6.878787878787879e-06, "loss": -0.2286, "num_tokens": 2313782.0, "reward": 0.5306993722915649, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.2777460217475891, "reward_meter_mean": 0.6844993233680725, "reward_meter_std": 0.36398857831954956, "reward_repeat_penalty_mean": 0.8611111044883728, "reward_repeat_penalty_std": 0.09848947077989578, "reward_std": 0.33074015378952026, "reward_total_composite_mean": 0.5306993722915649, "reward_total_composite_std": 0.33074015378952026, "reward_total_mean": 0.5306993722915649, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.2777460217475891, "rewards/meter/mean": 0.6844993233680725, "rewards/meter/std": 0.36398857831954956, "rewards/repeat_penalty/mean": 0.8611111044883728, "rewards/repeat_penalty/std": 0.09848947077989578, "rewards/total_composite/mean": 0.5306993722915649, "rewards/total_composite/std": 0.33074015378952026, "sampling/importance_sampling_ratio/max": 1.6992144584655762, "sampling/importance_sampling_ratio/mean": 1.0042911767959595, "sampling/importance_sampling_ratio/min": 0.24661383032798767, "sampling/sampling_logp_difference/max": 1.3999316692352295, "sampling/sampling_logp_difference/mean": 0.03939226269721985, "step": 1031 }, { "clip_ratio/high_max": 0.007109183439752087, "clip_ratio/high_mean": 0.007109183439752087, "clip_ratio/low_mean": 0.005716652231058106, "clip_ratio/low_min": 0.005716652231058106, "clip_ratio/region_mean": 0.012825835670810193, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 322.75, "completions/mean_terminated_length": 295.71429443359375, "completions/min_length": 279.0, "completions/min_terminated_length": 279.0, "entropy": 0.12301667220890522, "epoch": 0.04145077720207254, "frac_reward_zero_std": 0.0, "grad_norm": 1.4397751092910767, "learning_rate": 6.875757575757576e-06, "loss": -0.1879, "num_tokens": 2317508.0, "reward": 0.3974968194961548, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.2070196568965912, "reward_meter_mean": 0.8027582168579102, "reward_meter_std": 0.345333069562912, "reward_repeat_penalty_mean": 0.630974292755127, "reward_repeat_penalty_std": 0.19635772705078125, "reward_std": 0.23275509476661682, "reward_total_composite_mean": 0.3974968194961548, "reward_total_composite_std": 0.2327551245689392, "reward_total_mean": 0.3974968194961548, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.2070196568965912, "rewards/meter/mean": 0.8027582168579102, "rewards/meter/std": 0.345333069562912, "rewards/repeat_penalty/mean": 0.630974292755127, "rewards/repeat_penalty/std": 0.19635772705078125, "rewards/total_composite/mean": 0.3974968194961548, "rewards/total_composite/std": 0.2327551245689392, "sampling/importance_sampling_ratio/max": 1.966111660003662, "sampling/importance_sampling_ratio/mean": 1.0034023523330688, "sampling/importance_sampling_ratio/min": 0.04611998051404953, "sampling/sampling_logp_difference/max": 3.0765089988708496, "sampling/sampling_logp_difference/mean": 0.022777464240789413, "step": 1032 }, { "clip_ratio/high_max": 0.026730769779533148, "clip_ratio/high_mean": 0.026730769779533148, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/region_mean": 0.029089260380715132, "completions/clipped_ratio": 0.0, "completions/max_length": 53.0, "completions/max_terminated_length": 53.0, "completions/mean_length": 50.875, "completions/mean_terminated_length": 50.875, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.21698110736906528, "epoch": 0.041490942683857494, "frac_reward_zero_std": 0.0, "grad_norm": 16.486671447753906, "learning_rate": 6.872727272727273e-06, "loss": 0.0392, "num_tokens": 2319195.0, "reward": 0.8368604779243469, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8368604779243469, "reward_meter_std": 0.27409422397613525, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.27409422397613525, "reward_total_composite_mean": 0.8368604779243469, "reward_total_composite_std": 0.27409422397613525, "reward_total_mean": 0.8368604779243469, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8368604779243469, "rewards/meter/std": 0.27409422397613525, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8368604779243469, "rewards/total_composite/std": 0.27409422397613525, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001241683959961, "sampling/importance_sampling_ratio/min": 0.28702741861343384, "sampling/sampling_logp_difference/max": 1.2481775283813477, "sampling/sampling_logp_difference/mean": 0.04724201560020447, "step": 1033 }, { "clip_ratio/high_max": 0.022404347429983318, "clip_ratio/high_mean": 0.022404347429983318, "clip_ratio/low_mean": 0.02950768917798996, "clip_ratio/low_min": 0.02950768917798996, "clip_ratio/region_mean": 0.05191203660797328, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 67.625, "completions/mean_terminated_length": 67.625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.5639077946543694, "epoch": 0.04153110816564245, "frac_reward_zero_std": 0.0, "grad_norm": 5.985336780548096, "learning_rate": 6.869696969696971e-06, "loss": 0.003, "num_tokens": 2321016.0, "reward": 0.053763557225465775, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.053763557225465775, "reward_meter_std": 0.056538958102464676, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05653895437717438, "reward_total_composite_mean": 0.053763557225465775, "reward_total_composite_std": 0.056538958102464676, "reward_total_mean": 0.053763557225465775, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.053763557225465775, "rewards/meter/std": 0.056538958102464676, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.053763557225465775, "rewards/total_composite/std": 0.056538958102464676, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0089935064315796, "sampling/importance_sampling_ratio/min": 0.1506378948688507, "sampling/sampling_logp_difference/max": 1.892876386642456, "sampling/sampling_logp_difference/mean": 0.07932179421186447, "step": 1034 }, { "clip_ratio/high_max": 0.03407761920243502, "clip_ratio/high_mean": 0.03407761920243502, "clip_ratio/low_mean": 0.006497558439150453, "clip_ratio/low_min": 0.006497558439150453, "clip_ratio/region_mean": 0.04057517764158547, "completions/clipped_ratio": 0.0, "completions/max_length": 100.0, "completions/max_terminated_length": 100.0, "completions/mean_length": 95.875, "completions/mean_terminated_length": 95.875, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.18867953680455685, "epoch": 0.0415712736474274, "frac_reward_zero_std": 0.0, "grad_norm": 4.859825134277344, "learning_rate": 6.866666666666667e-06, "loss": 0.0078, "num_tokens": 2323151.0, "reward": 0.7163466215133667, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8145841360092163, "reward_meter_std": 0.33935561776161194, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.304019570350647, "reward_total_composite_mean": 0.7163466215133667, "reward_total_composite_std": 0.304019570350647, "reward_total_mean": 0.7163466215133667, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8145841360092163, "rewards/meter/std": 0.33935561776161194, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7163466215133667, "rewards/total_composite/std": 0.304019570350647, "sampling/importance_sampling_ratio/max": 1.6885980367660522, "sampling/importance_sampling_ratio/mean": 0.9907196164131165, "sampling/importance_sampling_ratio/min": 3.0297920422528435e-11, "sampling/sampling_logp_difference/max": 24.219942092895508, "sampling/sampling_logp_difference/mean": 0.08979999274015427, "step": 1035 }, { "clip_ratio/high_max": 0.021130952751263976, "clip_ratio/high_mean": 0.021130952751263976, "clip_ratio/low_mean": 0.021446609403938055, "clip_ratio/low_min": 0.021446609403938055, "clip_ratio/region_mean": 0.04257756215520203, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.875, "completions/mean_terminated_length": 34.875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.272716524079442, "epoch": 0.041611439129212356, "frac_reward_zero_std": 0.0, "grad_norm": 9.299543380737305, "learning_rate": 6.8636363636363645e-06, "loss": 0.0157, "num_tokens": 2324686.0, "reward": 0.9909951686859131, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9909951686859131, "reward_meter_std": 0.0066036987118422985, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006603698246181011, "reward_total_composite_mean": 0.9909951686859131, "reward_total_composite_std": 0.0066036987118422985, "reward_total_mean": 0.9909951686859131, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9909951686859131, "rewards/meter/std": 0.0066036987118422985, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9909951686859131, "rewards/total_composite/std": 0.0066036987118422985, "sampling/importance_sampling_ratio/max": 1.8759939670562744, "sampling/importance_sampling_ratio/mean": 1.0070772171020508, "sampling/importance_sampling_ratio/min": 0.49455809593200684, "sampling/sampling_logp_difference/max": 0.7040905952453613, "sampling/sampling_logp_difference/mean": 0.047876570373773575, "step": 1036 }, { "clip_ratio/high_max": 0.018199163547251374, "clip_ratio/high_mean": 0.018199163547251374, "clip_ratio/low_mean": 0.01997046614997089, "clip_ratio/low_min": 0.01997046614997089, "clip_ratio/region_mean": 0.03816962969722226, "completions/clipped_ratio": 0.0, "completions/max_length": 150.0, "completions/max_terminated_length": 150.0, "completions/mean_length": 133.875, "completions/mean_terminated_length": 133.875, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.2784906532615423, "epoch": 0.04165160461099731, "frac_reward_zero_std": 0.0, "grad_norm": 13.22099494934082, "learning_rate": 6.860606060606061e-06, "loss": -0.0406, "num_tokens": 2327237.0, "reward": 0.8550522923469543, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9935377836227417, "reward_meter_std": 0.007430072408169508, "reward_repeat_penalty_mean": 0.8857142925262451, "reward_repeat_penalty_std": 0.07324227690696716, "reward_std": 0.1225212812423706, "reward_total_composite_mean": 0.8550522923469543, "reward_total_composite_std": 0.1225212812423706, "reward_total_mean": 0.8550522923469543, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9935377836227417, "rewards/meter/std": 0.007430072408169508, "rewards/repeat_penalty/mean": 0.8857142925262451, "rewards/repeat_penalty/std": 0.07324227690696716, "rewards/total_composite/mean": 0.8550522923469543, "rewards/total_composite/std": 0.1225212812423706, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0073775053024292, "sampling/importance_sampling_ratio/min": 0.10639062523841858, "sampling/sampling_logp_difference/max": 2.24063777923584, "sampling/sampling_logp_difference/mean": 0.04958852007985115, "step": 1037 }, { "clip_ratio/high_max": 0.017664092825725675, "clip_ratio/high_mean": 0.017664092825725675, "clip_ratio/low_mean": 0.05519176600500941, "clip_ratio/low_min": 0.05519176600500941, "clip_ratio/region_mean": 0.07285585883073509, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 36.375, "completions/mean_terminated_length": 36.375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.8697552978992462, "epoch": 0.041691770092782264, "frac_reward_zero_std": 0.0, "grad_norm": 9.27010726928711, "learning_rate": 6.857575757575758e-06, "loss": 0.0376, "num_tokens": 2328768.0, "reward": 0.24713511765003204, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.24713511765003204, "reward_meter_std": 0.3414146900177002, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3414146900177002, "reward_total_composite_mean": 0.24713511765003204, "reward_total_composite_std": 0.3414146900177002, "reward_total_mean": 0.24713511765003204, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.24713511765003204, "rewards/meter/std": 0.3414146900177002, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.24713511765003204, "rewards/total_composite/std": 0.3414146900177002, "sampling/importance_sampling_ratio/max": 1.7845203876495361, "sampling/importance_sampling_ratio/mean": 1.0138072967529297, "sampling/importance_sampling_ratio/min": 0.3509252369403839, "sampling/sampling_logp_difference/max": 1.0471820831298828, "sampling/sampling_logp_difference/mean": 0.09440852701663971, "step": 1038 }, { "clip_ratio/high_max": 0.024171422701328993, "clip_ratio/high_mean": 0.024171422701328993, "clip_ratio/low_mean": 0.03564694127999246, "clip_ratio/low_min": 0.03564694127999246, "clip_ratio/region_mean": 0.059818363981321454, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 69.625, "completions/mean_terminated_length": 69.625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.47740989923477173, "epoch": 0.04173193557456722, "frac_reward_zero_std": 0.0, "grad_norm": 4.874062538146973, "learning_rate": 6.854545454545455e-06, "loss": 0.0252, "num_tokens": 2330669.0, "reward": 0.19316673278808594, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.19316673278808594, "reward_meter_std": 0.19894513487815857, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.19894511997699738, "reward_total_composite_mean": 0.19316673278808594, "reward_total_composite_std": 0.19894513487815857, "reward_total_mean": 0.19316673278808594, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.19316673278808594, "rewards/meter/std": 0.19894513487815857, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.19316673278808594, "rewards/total_composite/std": 0.19894513487815857, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0039423704147339, "sampling/importance_sampling_ratio/min": 0.2099989503622055, "sampling/sampling_logp_difference/max": 1.560652732849121, "sampling/sampling_logp_difference/mean": 0.0636194571852684, "step": 1039 }, { "clip_ratio/high_max": 0.025824516778811812, "clip_ratio/high_mean": 0.025824516778811812, "clip_ratio/low_mean": 0.016255452763289213, "clip_ratio/low_min": 0.016255452763289213, "clip_ratio/region_mean": 0.042079969542101026, "completions/clipped_ratio": 0.0, "completions/max_length": 182.0, "completions/max_terminated_length": 182.0, "completions/mean_length": 160.0, "completions/mean_terminated_length": 160.0, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.39750632736831903, "epoch": 0.04177210105635217, "frac_reward_zero_std": 0.0, "grad_norm": 5.022459030151367, "learning_rate": 6.851515151515153e-06, "loss": -0.0641, "num_tokens": 2333405.0, "reward": 0.6102761030197144, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.1035098284482956, "reward_meter_mean": 0.9006245136260986, "reward_meter_std": 0.21294330060482025, "reward_repeat_penalty_mean": 0.7440476417541504, "reward_repeat_penalty_std": 0.17643596231937408, "reward_std": 0.19687248766422272, "reward_total_composite_mean": 0.6102761030197144, "reward_total_composite_std": 0.19687247276306152, "reward_total_mean": 0.6102761030197144, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.1035098284482956, "rewards/meter/mean": 0.9006245136260986, "rewards/meter/std": 0.21294330060482025, "rewards/repeat_penalty/mean": 0.7440476417541504, "rewards/repeat_penalty/std": 0.17643596231937408, "rewards/total_composite/mean": 0.6102761030197144, "rewards/total_composite/std": 0.19687247276306152, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0185436010360718, "sampling/importance_sampling_ratio/min": 0.25236961245536804, "sampling/sampling_logp_difference/max": 1.3793213367462158, "sampling/sampling_logp_difference/mean": 0.05596721172332764, "step": 1040 }, { "clip_ratio/high_max": 0.012427689041942358, "clip_ratio/high_mean": 0.012427689041942358, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/region_mean": 0.015552689088508487, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 40.125, "completions/mean_terminated_length": 40.125, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.2032633163034916, "epoch": 0.041812266538137126, "frac_reward_zero_std": 0.0, "grad_norm": 16.1435604095459, "learning_rate": 6.848484848484849e-06, "loss": 0.0176, "num_tokens": 2334958.0, "reward": 0.9823716878890991, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9823716878890991, "reward_meter_std": 0.015363764949142933, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.015363766811788082, "reward_total_composite_mean": 0.9823716878890991, "reward_total_composite_std": 0.015363764949142933, "reward_total_mean": 0.9823716878890991, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9823716878890991, "rewards/meter/std": 0.015363764949142933, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9823716878890991, "rewards/total_composite/std": 0.015363764949142933, "sampling/importance_sampling_ratio/max": 1.6625901460647583, "sampling/importance_sampling_ratio/mean": 1.0097277164459229, "sampling/importance_sampling_ratio/min": 0.2506465017795563, "sampling/sampling_logp_difference/max": 1.3837116956710815, "sampling/sampling_logp_difference/mean": 0.03611617907881737, "step": 1041 }, { "clip_ratio/high_max": 0.023097628843970597, "clip_ratio/high_mean": 0.023097628843970597, "clip_ratio/low_mean": 0.007275132229551673, "clip_ratio/low_min": 0.007275132229551673, "clip_ratio/region_mean": 0.03037276107352227, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 126.5, "completions/mean_terminated_length": 126.5, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.1586785214021802, "epoch": 0.04185243201992208, "frac_reward_zero_std": 0.0, "grad_norm": 3.8245151042938232, "learning_rate": 6.845454545454546e-06, "loss": 0.0581, "num_tokens": 2337338.0, "reward": 0.5652507543563843, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7728732824325562, "reward_meter_std": 0.3359214961528778, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.10101525485515594, "reward_std": 0.22422267496585846, "reward_total_composite_mean": 0.5652507543563843, "reward_total_composite_std": 0.22422268986701965, "reward_total_mean": 0.5652507543563843, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7728732824325562, "rewards/meter/std": 0.3359214961528778, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.5652507543563843, "rewards/total_composite/std": 0.22422268986701965, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0046995878219604, "sampling/importance_sampling_ratio/min": 0.1569148302078247, "sampling/sampling_logp_difference/max": 1.852052092552185, "sampling/sampling_logp_difference/mean": 0.031546179205179214, "step": 1042 }, { "clip_ratio/high_max": 0.018411532044410706, "clip_ratio/high_mean": 0.018411532044410706, "clip_ratio/low_mean": 0.02028265898115933, "clip_ratio/low_min": 0.02028265898115933, "clip_ratio/region_mean": 0.038694191025570035, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.625, "completions/mean_terminated_length": 67.625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.3001182395964861, "epoch": 0.041892597501707034, "frac_reward_zero_std": 0.0, "grad_norm": 5.958546161651611, "learning_rate": 6.842424242424243e-06, "loss": 0.0128, "num_tokens": 2339119.0, "reward": 0.5403692722320557, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5403692722320557, "reward_meter_std": 0.4386324882507324, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.43863245844841003, "reward_total_composite_mean": 0.5403692722320557, "reward_total_composite_std": 0.4386324882507324, "reward_total_mean": 0.5403692722320557, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5403692722320557, "rewards/meter/std": 0.4386324882507324, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5403692722320557, "rewards/total_composite/std": 0.4386324882507324, "sampling/importance_sampling_ratio/max": 1.7478914260864258, "sampling/importance_sampling_ratio/mean": 1.0055934190750122, "sampling/importance_sampling_ratio/min": 0.19303473830223083, "sampling/sampling_logp_difference/max": 1.6448850631713867, "sampling/sampling_logp_difference/mean": 0.047015149146318436, "step": 1043 }, { "clip_ratio/high_max": 0.006274061510339379, "clip_ratio/high_mean": 0.006274061510339379, "clip_ratio/low_mean": 0.020476194215007126, "clip_ratio/low_min": 0.020476194215007126, "clip_ratio/region_mean": 0.026750255725346506, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 98.75, "completions/mean_terminated_length": 98.75, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.1880979547277093, "epoch": 0.04193276298349199, "frac_reward_zero_std": 0.0, "grad_norm": 3.107640266418457, "learning_rate": 6.83939393939394e-06, "loss": -0.0019, "num_tokens": 2341333.0, "reward": 0.4054219126701355, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4054219126701355, "reward_meter_std": 0.4075765311717987, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4075765013694763, "reward_total_composite_mean": 0.4054219126701355, "reward_total_composite_std": 0.4075765311717987, "reward_total_mean": 0.4054219126701355, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4054219126701355, "rewards/meter/std": 0.4075765311717987, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4054219126701355, "rewards/total_composite/std": 0.4075765311717987, "sampling/importance_sampling_ratio/max": 1.5978652238845825, "sampling/importance_sampling_ratio/mean": 1.0041404962539673, "sampling/importance_sampling_ratio/min": 0.40708449482917786, "sampling/sampling_logp_difference/max": 0.8987345695495605, "sampling/sampling_logp_difference/mean": 0.031202714890241623, "step": 1044 }, { "clip_ratio/high_max": 0.021053530042991042, "clip_ratio/high_mean": 0.021053530042991042, "clip_ratio/low_mean": 0.015083948150277138, "clip_ratio/low_min": 0.015083948150277138, "clip_ratio/region_mean": 0.03613747819326818, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 93.625, "completions/mean_terminated_length": 93.625, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.20175535418093204, "epoch": 0.04197292846527694, "frac_reward_zero_std": 0.0, "grad_norm": 6.213621616363525, "learning_rate": 6.8363636363636364e-06, "loss": -0.0095, "num_tokens": 2343378.0, "reward": 0.745398998260498, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.745398998260498, "reward_meter_std": 0.21442271769046783, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.21442271769046783, "reward_total_composite_mean": 0.745398998260498, "reward_total_composite_std": 0.21442271769046783, "reward_total_mean": 0.745398998260498, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.745398998260498, "rewards/meter/std": 0.21442271769046783, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.745398998260498, "rewards/total_composite/std": 0.21442271769046783, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0055217742919922, "sampling/importance_sampling_ratio/min": 0.35487717390060425, "sampling/sampling_logp_difference/max": 1.0359835624694824, "sampling/sampling_logp_difference/mean": 0.036333851516246796, "step": 1045 }, { "clip_ratio/high_max": 0.031451608054339886, "clip_ratio/high_mean": 0.031451608054339886, "clip_ratio/low_mean": 0.026830808725208044, "clip_ratio/low_min": 0.026830808725208044, "clip_ratio/region_mean": 0.05828241677954793, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 68.875, "completions/mean_terminated_length": 68.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.3025083411484957, "epoch": 0.042013093947061896, "frac_reward_zero_std": 0.0, "grad_norm": 8.28740406036377, "learning_rate": 6.833333333333334e-06, "loss": 0.0232, "num_tokens": 2345073.0, "reward": 0.5793697834014893, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5793697834014893, "reward_meter_std": 0.473821759223938, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.473821759223938, "reward_total_composite_mean": 0.5793697834014893, "reward_total_composite_std": 0.473821759223938, "reward_total_mean": 0.5793697834014893, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5793697834014893, "rewards/meter/std": 0.473821759223938, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5793697834014893, "rewards/total_composite/std": 0.473821759223938, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0050064325332642, "sampling/importance_sampling_ratio/min": 0.14157122373580933, "sampling/sampling_logp_difference/max": 1.9549522399902344, "sampling/sampling_logp_difference/mean": 0.058582451194524765, "step": 1046 }, { "clip_ratio/high_max": 0.0447791040642187, "clip_ratio/high_mean": 0.0447791040642187, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0447791040642187, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 139.0, "completions/mean_length": 181.0, "completions/mean_terminated_length": 133.71429443359375, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.30693456530570984, "epoch": 0.04205325942884685, "frac_reward_zero_std": 0.0, "grad_norm": 1.4744713306427002, "learning_rate": 6.83030303030303e-06, "loss": -0.2118, "num_tokens": 2347417.0, "reward": 0.7358262538909912, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.9518733024597168, "reward_meter_std": 0.08209281414747238, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.31345105171203613, "reward_total_composite_mean": 0.7358262538909912, "reward_total_composite_std": 0.31345105171203613, "reward_total_mean": 0.7358262538909912, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.9518733024597168, "rewards/meter/std": 0.08209281414747238, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7358262538909912, "rewards/total_composite/std": 0.31345105171203613, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017281770706177, "sampling/importance_sampling_ratio/min": 0.14382989704608917, "sampling/sampling_logp_difference/max": 1.9391239881515503, "sampling/sampling_logp_difference/mean": 0.05928022786974907, "step": 1047 }, { "clip_ratio/high_max": 0.026004510698840022, "clip_ratio/high_mean": 0.026004510698840022, "clip_ratio/low_mean": 0.014204545877873898, "clip_ratio/low_min": 0.014204545877873898, "clip_ratio/region_mean": 0.04020905657671392, "completions/clipped_ratio": 0.0, "completions/max_length": 50.0, "completions/max_terminated_length": 50.0, "completions/mean_length": 47.125, "completions/mean_terminated_length": 47.125, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.374115826562047, "epoch": 0.0420934249106318, "frac_reward_zero_std": 0.0, "grad_norm": 8.561767578125, "learning_rate": 6.827272727272728e-06, "loss": -0.0116, "num_tokens": 2349002.0, "reward": 0.6955418586730957, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6955418586730957, "reward_meter_std": 0.4201207160949707, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4201207160949707, "reward_total_composite_mean": 0.6955418586730957, "reward_total_composite_std": 0.4201207160949707, "reward_total_mean": 0.6955418586730957, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6955418586730957, "rewards/meter/std": 0.4201207160949707, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6955418586730957, "rewards/total_composite/std": 0.4201207160949707, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.009395718574524, "sampling/importance_sampling_ratio/min": 0.11048772186040878, "sampling/sampling_logp_difference/max": 2.202850818634033, "sampling/sampling_logp_difference/mean": 0.051052458584308624, "step": 1048 }, { "clip_ratio/high_max": 0.012093263794668019, "clip_ratio/high_mean": 0.012093263794668019, "clip_ratio/low_mean": 0.012227151426486671, "clip_ratio/low_min": 0.012227151426486671, "clip_ratio/region_mean": 0.02432041522115469, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 144.75, "completions/mean_terminated_length": 144.75, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.1037906464189291, "epoch": 0.04213359039241676, "frac_reward_zero_std": 0.0, "grad_norm": 8.349677085876465, "learning_rate": 6.824242424242425e-06, "loss": 0.0138, "num_tokens": 2351424.0, "reward": 0.6298381090164185, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.716080904006958, "reward_meter_std": 0.3239102363586426, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.2847732603549957, "reward_total_composite_mean": 0.6298381090164185, "reward_total_composite_std": 0.2847732603549957, "reward_total_mean": 0.6298381090164185, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.716080904006958, "rewards/meter/std": 0.3239102363586426, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6298381090164185, "rewards/total_composite/std": 0.2847732603549957, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015819072723389, "sampling/importance_sampling_ratio/min": 0.2912403643131256, "sampling/sampling_logp_difference/max": 1.2336063385009766, "sampling/sampling_logp_difference/mean": 0.020672447979450226, "step": 1049 }, { "clip_ratio/high_max": 0.020683051785454154, "clip_ratio/high_mean": 0.020683051785454154, "clip_ratio/low_mean": 0.008656773250550032, "clip_ratio/low_min": 0.008656773250550032, "clip_ratio/region_mean": 0.029339825036004186, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 72.875, "completions/mean_terminated_length": 72.875, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.3899807743728161, "epoch": 0.04217375587420171, "frac_reward_zero_std": 0.0, "grad_norm": 9.696529388427734, "learning_rate": 6.821212121212122e-06, "loss": 0.0067, "num_tokens": 2353303.0, "reward": 0.7290558815002441, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7290558815002441, "reward_meter_std": 0.3074507713317871, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3074507415294647, "reward_total_composite_mean": 0.7290558815002441, "reward_total_composite_std": 0.3074507713317871, "reward_total_mean": 0.7290558815002441, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7290558815002441, "rewards/meter/std": 0.3074507713317871, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7290558815002441, "rewards/total_composite/std": 0.3074507713317871, "sampling/importance_sampling_ratio/max": 1.8488932847976685, "sampling/importance_sampling_ratio/mean": 0.9999740719795227, "sampling/importance_sampling_ratio/min": 0.02225572057068348, "sampling/sampling_logp_difference/max": 3.8051562309265137, "sampling/sampling_logp_difference/mean": 0.07516960054636002, "step": 1050 }, { "epoch": 0.04217375587420171, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/max_length": 395.0769230769231, "eval_completions/max_terminated_length": 383.3076923076923, "eval_completions/mean_length": 226.30769230769232, "eval_completions/mean_terminated_length": 223.2651108961839, "eval_completions/min_length": 73.84615384615384, "eval_completions/min_terminated_length": 73.84615384615384, "eval_entropy": 0.13593376485201028, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 2353303.0, "eval_reward": 0.33965545204969555, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.8679342178198007, "eval_reward_count_adherence_std": 0.11943456530570984, "eval_reward_meter_mean": 0.5502180938537304, "eval_reward_meter_std": 0.40295018599583554, "eval_reward_repeat_penalty_mean": 0.724188873401055, "eval_reward_repeat_penalty_std": 0.2071983149418464, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.33965545204969555, "eval_reward_total_composite_std": 0.3024227091899285, "eval_reward_total_mean": 0.33965545204969555, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.8679342178198007, "eval_rewards/count_adherence/std": 0.11943456530570984, "eval_rewards/meter/mean": 0.5502180938537304, "eval_rewards/meter/std": 0.40295018599583554, "eval_rewards/repeat_penalty/mean": 0.724188873401055, "eval_rewards/repeat_penalty/std": 0.2071983149418464, "eval_rewards/total_composite/mean": 0.33965545204969555, "eval_rewards/total_composite/std": 0.3024227091899285, "eval_runtime": 74.9052, "eval_samples_per_second": 1.388, "eval_sampling/importance_sampling_ratio/max": 1.4160706263322096, "eval_sampling/importance_sampling_ratio/mean": 1.0033980562136724, "eval_sampling/importance_sampling_ratio/min": 0.3627954079554631, "eval_sampling/sampling_logp_difference/max": 1.0244657076322115, "eval_sampling/sampling_logp_difference/mean": 0.015186995840989627, "eval_steps_per_second": 0.174, "step": 1050 }, { "clip_ratio/high_max": 0.026332546956837177, "clip_ratio/high_mean": 0.026332546956837177, "clip_ratio/low_mean": 0.005197861581109464, "clip_ratio/low_min": 0.005197861581109464, "clip_ratio/region_mean": 0.03153040853794664, "completions/clipped_ratio": 0.0, "completions/max_length": 235.0, "completions/max_terminated_length": 235.0, "completions/mean_length": 212.375, "completions/mean_terminated_length": 212.375, "completions/min_length": 183.0, "completions/min_terminated_length": 183.0, "entropy": 0.31745208986103535, "epoch": 0.042213921355986665, "frac_reward_zero_std": 0.0, "grad_norm": 2.1154820919036865, "learning_rate": 6.818181818181818e-06, "loss": 0.0139, "num_tokens": 2356570.0, "reward": 0.5123095512390137, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.78125, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.802075982093811, "reward_meter_std": 0.3460130989551544, "reward_repeat_penalty_mean": 0.7899305820465088, "reward_repeat_penalty_std": 0.090753473341465, "reward_std": 0.2536678910255432, "reward_total_composite_mean": 0.5123095512390137, "reward_total_composite_std": 0.2536679208278656, "reward_total_mean": 0.5123095512390137, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.78125, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.802075982093811, "rewards/meter/std": 0.3460130989551544, "rewards/repeat_penalty/mean": 0.7899305820465088, "rewards/repeat_penalty/std": 0.090753473341465, "rewards/total_composite/mean": 0.5123095512390137, "rewards/total_composite/std": 0.2536679208278656, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0035831928253174, "sampling/importance_sampling_ratio/min": 0.21022215485572815, "sampling/sampling_logp_difference/max": 1.5595903396606445, "sampling/sampling_logp_difference/mean": 0.038277287036180496, "step": 1051 }, { "clip_ratio/high_max": 0.05841635470278561, "clip_ratio/high_mean": 0.05841635470278561, "clip_ratio/low_mean": 0.0028735632076859474, "clip_ratio/low_min": 0.0028735632076859474, "clip_ratio/region_mean": 0.06128991791047156, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 89.75, "completions/mean_terminated_length": 89.75, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.586686696857214, "epoch": 0.04225408683777162, "frac_reward_zero_std": 0.0, "grad_norm": 4.994956970214844, "learning_rate": 6.8151515151515155e-06, "loss": -0.0141, "num_tokens": 2358608.0, "reward": 0.9309325218200684, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9309325218200684, "reward_meter_std": 0.17771661281585693, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.17771658301353455, "reward_total_composite_mean": 0.9309325218200684, "reward_total_composite_std": 0.17771661281585693, "reward_total_mean": 0.9309325218200684, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9309325218200684, "rewards/meter/std": 0.17771661281585693, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9309325218200684, "rewards/total_composite/std": 0.17771661281585693, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0161553621292114, "sampling/importance_sampling_ratio/min": 0.2512664496898651, "sampling/sampling_logp_difference/max": 1.3812413215637207, "sampling/sampling_logp_difference/mean": 0.07444296777248383, "step": 1052 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.01263241225387901, "clip_ratio/low_min": 0.01263241225387901, "clip_ratio/region_mean": 0.016363755450583994, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.25, "completions/mean_terminated_length": 69.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.13288897648453712, "epoch": 0.04229425231955657, "frac_reward_zero_std": 0.0, "grad_norm": 4.2257232666015625, "learning_rate": 6.812121212121212e-06, "loss": 0.0063, "num_tokens": 2360466.0, "reward": 0.9586122035980225, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9586122035980225, "reward_meter_std": 0.00875465851277113, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00875465851277113, "reward_total_composite_mean": 0.9586122035980225, "reward_total_composite_std": 0.00875465851277113, "reward_total_mean": 0.9586122035980225, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9586122035980225, "rewards/meter/std": 0.00875465851277113, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9586122035980225, "rewards/total_composite/std": 0.00875465851277113, "sampling/importance_sampling_ratio/max": 1.3229237794876099, "sampling/importance_sampling_ratio/mean": 0.9992114901542664, "sampling/importance_sampling_ratio/min": 0.3453146815299988, "sampling/sampling_logp_difference/max": 1.0632991790771484, "sampling/sampling_logp_difference/mean": 0.020308073610067368, "step": 1053 }, { "clip_ratio/high_max": 0.02460848237387836, "clip_ratio/high_mean": 0.02460848237387836, "clip_ratio/low_mean": 0.008994319243356586, "clip_ratio/low_min": 0.008994319243356586, "clip_ratio/region_mean": 0.033602801617234945, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 96.875, "completions/mean_terminated_length": 96.875, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.17869562469422817, "epoch": 0.04233441780134153, "frac_reward_zero_std": 0.0, "grad_norm": 2.932860851287842, "learning_rate": 6.80909090909091e-06, "loss": 0.0069, "num_tokens": 2362481.0, "reward": 0.9002121686935425, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9400673508644104, "reward_meter_std": 0.03255482763051987, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11076170206069946, "reward_total_composite_mean": 0.9002121686935425, "reward_total_composite_std": 0.11076170206069946, "reward_total_mean": 0.9002121686935425, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9400673508644104, "rewards/meter/std": 0.03255482763051987, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9002121686935425, "rewards/total_composite/std": 0.11076170206069946, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9991159439086914, "sampling/importance_sampling_ratio/min": 0.20014327764511108, "sampling/sampling_logp_difference/max": 1.6087217330932617, "sampling/sampling_logp_difference/mean": 0.0384301133453846, "step": 1054 }, { "clip_ratio/high_max": 0.01200975279789418, "clip_ratio/high_mean": 0.01200975279789418, "clip_ratio/low_mean": 0.014537818904500455, "clip_ratio/low_min": 0.014537818904500455, "clip_ratio/region_mean": 0.026547571702394634, "completions/clipped_ratio": 0.0, "completions/max_length": 156.0, "completions/max_terminated_length": 156.0, "completions/mean_length": 137.875, "completions/mean_terminated_length": 137.875, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.4246663171797991, "epoch": 0.04237458328312648, "frac_reward_zero_std": 0.0, "grad_norm": 3.7541143894195557, "learning_rate": 6.806060606060607e-06, "loss": 0.1006, "num_tokens": 2364800.0, "reward": 0.14893272519111633, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7083333730697632, "reward_count_adherence_std": 0.11785111576318741, "reward_meter_mean": 0.25745514035224915, "reward_meter_std": 0.2247081845998764, "reward_repeat_penalty_mean": 0.6964285969734192, "reward_repeat_penalty_std": 0.2466983199119568, "reward_std": 0.16499118506908417, "reward_total_composite_mean": 0.14893272519111633, "reward_total_composite_std": 0.16499118506908417, "reward_total_mean": 0.14893272519111633, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7083333730697632, "rewards/count_adherence/std": 0.11785111576318741, "rewards/meter/mean": 0.25745514035224915, "rewards/meter/std": 0.2247081845998764, "rewards/repeat_penalty/mean": 0.6964285969734192, "rewards/repeat_penalty/std": 0.2466983199119568, "rewards/total_composite/mean": 0.14893272519111633, "rewards/total_composite/std": 0.16499118506908417, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0185996294021606, "sampling/importance_sampling_ratio/min": 0.16175222396850586, "sampling/sampling_logp_difference/max": 1.8216896057128906, "sampling/sampling_logp_difference/mean": 0.05157129094004631, "step": 1055 }, { "clip_ratio/high_max": 0.02171629574149847, "clip_ratio/high_mean": 0.02171629574149847, "clip_ratio/low_mean": 0.019135613925755024, "clip_ratio/low_min": 0.019135613925755024, "clip_ratio/region_mean": 0.040851909667253494, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 70.75, "completions/mean_terminated_length": 70.75, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.17993433214724064, "epoch": 0.042414748764911435, "frac_reward_zero_std": 0.0, "grad_norm": 12.402931213378906, "learning_rate": 6.803030303030304e-06, "loss": 0.0447, "num_tokens": 2366750.0, "reward": 0.30388349294662476, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.30388349294662476, "reward_meter_std": 0.2288217395544052, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22882172465324402, "reward_total_composite_mean": 0.30388349294662476, "reward_total_composite_std": 0.2288217395544052, "reward_total_mean": 0.30388349294662476, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.30388349294662476, "rewards/meter/std": 0.2288217395544052, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.30388349294662476, "rewards/total_composite/std": 0.2288217395544052, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9882052540779114, "sampling/importance_sampling_ratio/min": 0.07319188117980957, "sampling/sampling_logp_difference/max": 2.614670753479004, "sampling/sampling_logp_difference/mean": 0.05324329808354378, "step": 1056 }, { "clip_ratio/high_max": 0.0039429860189557076, "clip_ratio/high_mean": 0.0039429860189557076, "clip_ratio/low_mean": 0.013276972225867212, "clip_ratio/low_min": 0.013276972225867212, "clip_ratio/region_mean": 0.01721995824482292, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 349.0, "completions/mean_terminated_length": 349.0, "completions/min_length": 337.0, "completions/min_terminated_length": 337.0, "entropy": 0.14950765296816826, "epoch": 0.04245491424669639, "frac_reward_zero_std": 0.0, "grad_norm": 1.8229259252548218, "learning_rate": 6.800000000000001e-06, "loss": 0.0021, "num_tokens": 2371070.0, "reward": 0.14668160676956177, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.762499988079071, "reward_count_adherence_std": 0.05175492912530899, "reward_meter_mean": 0.3180350661277771, "reward_meter_std": 0.4220852255821228, "reward_repeat_penalty_mean": 0.6007440090179443, "reward_repeat_penalty_std": 0.11201495677232742, "reward_std": 0.197221577167511, "reward_total_composite_mean": 0.14668160676956177, "reward_total_composite_std": 0.19722160696983337, "reward_total_mean": 0.14668160676956177, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.762499988079071, "rewards/count_adherence/std": 0.05175492912530899, "rewards/meter/mean": 0.3180350661277771, "rewards/meter/std": 0.4220852255821228, "rewards/repeat_penalty/mean": 0.6007440090179443, "rewards/repeat_penalty/std": 0.11201495677232742, "rewards/total_composite/mean": 0.14668160676956177, "rewards/total_composite/std": 0.19722160696983337, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001380205154419, "sampling/importance_sampling_ratio/min": 0.18453148007392883, "sampling/sampling_logp_difference/max": 1.6899352073669434, "sampling/sampling_logp_difference/mean": 0.020703818649053574, "step": 1057 }, { "clip_ratio/high_max": 0.0041363348718732595, "clip_ratio/high_mean": 0.0041363348718732595, "clip_ratio/low_mean": 0.008206001890357584, "clip_ratio/low_min": 0.008206001890357584, "clip_ratio/region_mean": 0.012342336762230843, "completions/clipped_ratio": 0.0, "completions/max_length": 367.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 347.625, "completions/mean_terminated_length": 347.625, "completions/min_length": 326.0, "completions/min_terminated_length": 326.0, "entropy": 0.07590485410764813, "epoch": 0.04249507972848134, "frac_reward_zero_std": 0.0, "grad_norm": 2.157697916030884, "learning_rate": 6.796969696969697e-06, "loss": -0.0193, "num_tokens": 2375539.0, "reward": 0.26854199171066284, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7222222089767456, "reward_count_adherence_std": 0.059391383081674576, "reward_meter_mean": 0.7610747814178467, "reward_meter_std": 0.25479352474212646, "reward_repeat_penalty_mean": 0.4871794581413269, "reward_repeat_penalty_std": 0.16510966420173645, "reward_std": 0.14636212587356567, "reward_total_composite_mean": 0.26854199171066284, "reward_total_composite_std": 0.14636212587356567, "reward_total_mean": 0.26854199171066284, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7222222089767456, "rewards/count_adherence/std": 0.059391383081674576, "rewards/meter/mean": 0.7610747814178467, "rewards/meter/std": 0.25479352474212646, "rewards/repeat_penalty/mean": 0.4871794581413269, "rewards/repeat_penalty/std": 0.16510966420173645, "rewards/total_composite/mean": 0.26854199171066284, "rewards/total_composite/std": 0.14636212587356567, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0041615962982178, "sampling/importance_sampling_ratio/min": 0.12113604694604874, "sampling/sampling_logp_difference/max": 2.1108410358428955, "sampling/sampling_logp_difference/mean": 0.017443938180804253, "step": 1058 }, { "clip_ratio/high_max": 0.006493506487458944, "clip_ratio/high_mean": 0.006493506487458944, "clip_ratio/low_mean": 0.02205127221532166, "clip_ratio/low_min": 0.02205127221532166, "clip_ratio/region_mean": 0.028544778702780604, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.32050950825214386, "epoch": 0.0425352452102663, "frac_reward_zero_std": 0.0, "grad_norm": 3.8445513248443604, "learning_rate": 6.793939393939395e-06, "loss": 0.0103, "num_tokens": 2377463.0, "reward": 0.1600387543439865, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.1600387543439865, "reward_meter_std": 0.3069753348827362, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3069753646850586, "reward_total_composite_mean": 0.1600387543439865, "reward_total_composite_std": 0.3069753348827362, "reward_total_mean": 0.1600387543439865, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.1600387543439865, "rewards/meter/std": 0.3069753348827362, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.1600387543439865, "rewards/total_composite/std": 0.3069753348827362, "sampling/importance_sampling_ratio/max": 1.8910245895385742, "sampling/importance_sampling_ratio/mean": 1.008049726486206, "sampling/importance_sampling_ratio/min": 0.20700837671756744, "sampling/sampling_logp_difference/max": 1.574995994567871, "sampling/sampling_logp_difference/mean": 0.0510094054043293, "step": 1059 }, { "clip_ratio/high_max": 0.01580244250362739, "clip_ratio/high_mean": 0.01580244250362739, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.01580244250362739, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 367.5, "completions/mean_terminated_length": 319.3333435058594, "completions/min_length": 299.0, "completions/min_terminated_length": 299.0, "entropy": 0.1975250532850623, "epoch": 0.04257541069205125, "frac_reward_zero_std": 0.0, "grad_norm": 1.0387495756149292, "learning_rate": 6.790909090909091e-06, "loss": -0.3504, "num_tokens": 2381179.0, "reward": 0.3815041184425354, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.7613636255264282, "reward_count_adherence_std": 0.12798961997032166, "reward_meter_mean": 0.8225868344306946, "reward_meter_std": 0.18334050476551056, "reward_repeat_penalty_mean": 0.7348039150238037, "reward_repeat_penalty_std": 0.1173701137304306, "reward_std": 0.23795180022716522, "reward_total_composite_mean": 0.3815041184425354, "reward_total_composite_std": 0.23795181512832642, "reward_total_mean": 0.3815041184425354, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.7613636255264282, "rewards/count_adherence/std": 0.12798961997032166, "rewards/meter/mean": 0.8225868344306946, "rewards/meter/std": 0.18334050476551056, "rewards/repeat_penalty/mean": 0.7348039150238037, "rewards/repeat_penalty/std": 0.1173701137304306, "rewards/total_composite/mean": 0.3815041184425354, "rewards/total_composite/std": 0.23795181512832642, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0062330961227417, "sampling/importance_sampling_ratio/min": 0.16153423488140106, "sampling/sampling_logp_difference/max": 1.8230382204055786, "sampling/sampling_logp_difference/mean": 0.028801359236240387, "step": 1060 }, { "clip_ratio/high_max": 0.01467954705003649, "clip_ratio/high_mean": 0.01467954705003649, "clip_ratio/low_mean": 0.015588806010782719, "clip_ratio/low_min": 0.015588806010782719, "clip_ratio/region_mean": 0.03026835306081921, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 300.375, "completions/mean_terminated_length": 300.375, "completions/min_length": 289.0, "completions/min_terminated_length": 289.0, "entropy": 0.21212731674313545, "epoch": 0.042615576173836205, "frac_reward_zero_std": 0.0, "grad_norm": 2.5813870429992676, "learning_rate": 6.787878787878789e-06, "loss": -0.0089, "num_tokens": 2385246.0, "reward": 0.4012376070022583, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8928571939468384, "reward_count_adherence_std": 0.06613000482320786, "reward_meter_mean": 0.560438871383667, "reward_meter_std": 0.4554891288280487, "reward_repeat_penalty_mean": 0.7440122365951538, "reward_repeat_penalty_std": 0.10529463738203049, "reward_std": 0.3423055112361908, "reward_total_composite_mean": 0.4012376070022583, "reward_total_composite_std": 0.3423055410385132, "reward_total_mean": 0.4012376070022583, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8928571939468384, "rewards/count_adherence/std": 0.06613000482320786, "rewards/meter/mean": 0.560438871383667, "rewards/meter/std": 0.4554891288280487, "rewards/repeat_penalty/mean": 0.7440122365951538, "rewards/repeat_penalty/std": 0.10529463738203049, "rewards/total_composite/mean": 0.4012376070022583, "rewards/total_composite/std": 0.3423055410385132, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005553126335144, "sampling/importance_sampling_ratio/min": 0.18615710735321045, "sampling/sampling_logp_difference/max": 1.681164264678955, "sampling/sampling_logp_difference/mean": 0.032200075685977936, "step": 1061 }, { "clip_ratio/high_max": 0.004816017230041325, "clip_ratio/high_mean": 0.004816017230041325, "clip_ratio/low_mean": 0.016825482714921236, "clip_ratio/low_min": 0.016825482714921236, "clip_ratio/region_mean": 0.02164149994496256, "completions/clipped_ratio": 0.0, "completions/max_length": 207.0, "completions/max_terminated_length": 207.0, "completions/mean_length": 177.375, "completions/mean_terminated_length": 177.375, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.1391224949620664, "epoch": 0.04265574165562116, "frac_reward_zero_std": 0.0, "grad_norm": 2.7553000450134277, "learning_rate": 6.7848484848484855e-06, "loss": 0.0723, "num_tokens": 2387921.0, "reward": 0.1175825372338295, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3804289400577545, "reward_meter_std": 0.4567885398864746, "reward_repeat_penalty_mean": 0.5972222089767456, "reward_repeat_penalty_std": 0.25845491886138916, "reward_std": 0.16052624583244324, "reward_total_composite_mean": 0.1175825372338295, "reward_total_composite_std": 0.16052623093128204, "reward_total_mean": 0.1175825372338295, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3804289400577545, "rewards/meter/std": 0.4567885398864746, "rewards/repeat_penalty/mean": 0.5972222089767456, "rewards/repeat_penalty/std": 0.25845491886138916, "rewards/total_composite/mean": 0.1175825372338295, "rewards/total_composite/std": 0.16052623093128204, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005418062210083, "sampling/importance_sampling_ratio/min": 0.34907352924346924, "sampling/sampling_logp_difference/max": 1.2727034091949463, "sampling/sampling_logp_difference/mean": 0.02786877565085888, "step": 1062 }, { "clip_ratio/high_max": 0.02170370938256383, "clip_ratio/high_mean": 0.02170370938256383, "clip_ratio/low_mean": 0.014632937265560031, "clip_ratio/low_min": 0.014632937265560031, "clip_ratio/region_mean": 0.03633664664812386, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.15965153090655804, "epoch": 0.04269590713740611, "frac_reward_zero_std": 0.0, "grad_norm": 7.346792697906494, "learning_rate": 6.781818181818183e-06, "loss": 0.0052, "num_tokens": 2389705.0, "reward": 0.4735042452812195, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4735042452812195, "reward_meter_std": 0.4612884223461151, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4612884521484375, "reward_total_composite_mean": 0.4735042452812195, "reward_total_composite_std": 0.4612884223461151, "reward_total_mean": 0.4735042452812195, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4735042452812195, "rewards/meter/std": 0.4612884223461151, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4735042452812195, "rewards/total_composite/std": 0.4612884223461151, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0069684982299805, "sampling/importance_sampling_ratio/min": 0.3388230502605438, "sampling/sampling_logp_difference/max": 1.0822772979736328, "sampling/sampling_logp_difference/mean": 0.04212275519967079, "step": 1063 }, { "clip_ratio/high_max": 0.002241053676698357, "clip_ratio/high_mean": 0.002241053676698357, "clip_ratio/low_mean": 0.005747126415371895, "clip_ratio/low_min": 0.005747126415371895, "clip_ratio/region_mean": 0.007988180092070252, "completions/clipped_ratio": 0.0, "completions/max_length": 174.0, "completions/max_terminated_length": 174.0, "completions/mean_length": 167.875, "completions/mean_terminated_length": 167.875, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.15179980965331197, "epoch": 0.042736072619191066, "frac_reward_zero_std": 0.0, "grad_norm": 2.248720407485962, "learning_rate": 6.778787878787879e-06, "loss": 0.0144, "num_tokens": 2392392.0, "reward": 0.4253278374671936, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7978705167770386, "reward_meter_std": 0.26100829243659973, "reward_repeat_penalty_mean": 0.7222222089767456, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.12737736105918884, "reward_total_composite_mean": 0.4253278374671936, "reward_total_composite_std": 0.12737736105918884, "reward_total_mean": 0.4253278374671936, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7978705167770386, "rewards/meter/std": 0.26100829243659973, "rewards/repeat_penalty/mean": 0.7222222089767456, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.4253278374671936, "rewards/total_composite/std": 0.12737736105918884, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0040228366851807, "sampling/importance_sampling_ratio/min": 0.006912579294294119, "sampling/sampling_logp_difference/max": 4.974412441253662, "sampling/sampling_logp_difference/mean": 0.026998618617653847, "step": 1064 }, { "clip_ratio/high_max": 0.014428413240239024, "clip_ratio/high_mean": 0.014428413240239024, "clip_ratio/low_mean": 0.02187028736807406, "clip_ratio/low_min": 0.02187028736807406, "clip_ratio/region_mean": 0.036298700608313084, "completions/clipped_ratio": 0.0, "completions/max_length": 53.0, "completions/max_terminated_length": 53.0, "completions/mean_length": 51.375, "completions/mean_terminated_length": 51.375, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.17064102366566658, "epoch": 0.04277623810097602, "frac_reward_zero_std": 0.0, "grad_norm": 7.699649810791016, "learning_rate": 6.7757575757575765e-06, "loss": 0.0031, "num_tokens": 2393979.0, "reward": 0.3520212769508362, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3520212769508362, "reward_meter_std": 0.2575727701187134, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2575727701187134, "reward_total_composite_mean": 0.3520212769508362, "reward_total_composite_std": 0.2575727701187134, "reward_total_mean": 0.3520212769508362, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3520212769508362, "rewards/meter/std": 0.2575727701187134, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3520212769508362, "rewards/total_composite/std": 0.2575727701187134, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0043641328811646, "sampling/importance_sampling_ratio/min": 0.34641963243484497, "sampling/sampling_logp_difference/max": 1.1649155616760254, "sampling/sampling_logp_difference/mean": 0.03394267335534096, "step": 1065 }, { "clip_ratio/high_max": 0.022364818840287626, "clip_ratio/high_mean": 0.022364818840287626, "clip_ratio/low_mean": 0.00916612590663135, "clip_ratio/low_min": 0.00916612590663135, "clip_ratio/region_mean": 0.031530944746918976, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.25, "completions/mean_terminated_length": 67.25, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23792543355375528, "epoch": 0.042816403582760974, "frac_reward_zero_std": 0.0, "grad_norm": 4.696002006530762, "learning_rate": 6.772727272727273e-06, "loss": 0.0134, "num_tokens": 2395725.0, "reward": 0.7684643864631653, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7684643864631653, "reward_meter_std": 0.3259983956813812, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3259983956813812, "reward_total_composite_mean": 0.7684643864631653, "reward_total_composite_std": 0.3259983956813812, "reward_total_mean": 0.7684643864631653, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7684643864631653, "rewards/meter/std": 0.3259983956813812, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7684643864631653, "rewards/total_composite/std": 0.3259983956813812, "sampling/importance_sampling_ratio/max": 1.9387584924697876, "sampling/importance_sampling_ratio/mean": 0.9985710978507996, "sampling/importance_sampling_ratio/min": 0.22107702493667603, "sampling/sampling_logp_difference/max": 1.5092440843582153, "sampling/sampling_logp_difference/mean": 0.04000062495470047, "step": 1066 }, { "clip_ratio/high_max": 0.01831344375386834, "clip_ratio/high_mean": 0.01831344375386834, "clip_ratio/low_mean": 0.010416666977107525, "clip_ratio/low_min": 0.010416666977107525, "clip_ratio/region_mean": 0.028730110730975866, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 35.25, "completions/mean_terminated_length": 35.25, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.2819352522492409, "epoch": 0.04285656906454593, "frac_reward_zero_std": 0.0, "grad_norm": 10.04726791381836, "learning_rate": 6.76969696969697e-06, "loss": 0.0273, "num_tokens": 2397159.0, "reward": 0.40233293175697327, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.40233293175697327, "reward_meter_std": 0.35556212067604065, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35556209087371826, "reward_total_composite_mean": 0.40233293175697327, "reward_total_composite_std": 0.35556212067604065, "reward_total_mean": 0.40233293175697327, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.40233293175697327, "rewards/meter/std": 0.35556212067604065, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.40233293175697327, "rewards/total_composite/std": 0.35556212067604065, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9973819255828857, "sampling/importance_sampling_ratio/min": 0.32253700494766235, "sampling/sampling_logp_difference/max": 1.1315374374389648, "sampling/sampling_logp_difference/mean": 0.0461297444999218, "step": 1067 }, { "clip_ratio/high_max": 0.02335231169126928, "clip_ratio/high_mean": 0.02335231169126928, "clip_ratio/low_mean": 0.0076530613005161285, "clip_ratio/low_min": 0.0076530613005161285, "clip_ratio/region_mean": 0.031005372991785407, "completions/clipped_ratio": 0.0, "completions/max_length": 49.0, "completions/max_terminated_length": 49.0, "completions/mean_length": 47.375, "completions/mean_terminated_length": 47.375, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.15641734562814236, "epoch": 0.04289673454633088, "frac_reward_zero_std": 0.0, "grad_norm": 3.635596752166748, "learning_rate": 6.7666666666666665e-06, "loss": 0.0118, "num_tokens": 2398762.0, "reward": 0.7638416290283203, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7638416290283203, "reward_meter_std": 0.2724026143550873, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2724026143550873, "reward_total_composite_mean": 0.7638416290283203, "reward_total_composite_std": 0.2724026143550873, "reward_total_mean": 0.7638416290283203, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7638416290283203, "rewards/meter/std": 0.2724026143550873, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7638416290283203, "rewards/total_composite/std": 0.2724026143550873, "sampling/importance_sampling_ratio/max": 1.4794321060180664, "sampling/importance_sampling_ratio/mean": 1.0070512294769287, "sampling/importance_sampling_ratio/min": 0.33194637298583984, "sampling/sampling_logp_difference/max": 1.1027817726135254, "sampling/sampling_logp_difference/mean": 0.026672368869185448, "step": 1068 }, { "clip_ratio/high_max": 0.009160695597529411, "clip_ratio/high_mean": 0.009160695597529411, "clip_ratio/low_mean": 0.007008010521531105, "clip_ratio/low_min": 0.007008010521531105, "clip_ratio/region_mean": 0.016168706119060516, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 370.0, "completions/mean_length": 370.25, "completions/mean_terminated_length": 350.0000305175781, "completions/min_length": 328.0, "completions/min_terminated_length": 328.0, "entropy": 0.11531824059784412, "epoch": 0.042936900028115836, "frac_reward_zero_std": 0.0, "grad_norm": 1.4912827014923096, "learning_rate": 6.763636363636365e-06, "loss": -0.1635, "num_tokens": 2403084.0, "reward": 0.10642074048519135, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.7291666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.26017892360687256, "reward_meter_std": 0.2312571257352829, "reward_repeat_penalty_mean": 0.6733911633491516, "reward_repeat_penalty_std": 0.18306629359722137, "reward_std": 0.06904293596744537, "reward_total_composite_mean": 0.10642074048519135, "reward_total_composite_std": 0.06904294341802597, "reward_total_mean": 0.10642074048519135, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.7291666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.26017892360687256, "rewards/meter/std": 0.2312571257352829, "rewards/repeat_penalty/mean": 0.6733911633491516, "rewards/repeat_penalty/std": 0.18306629359722137, "rewards/total_composite/mean": 0.10642074048519135, "rewards/total_composite/std": 0.06904294341802597, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0042506456375122, "sampling/importance_sampling_ratio/min": 0.15768760442733765, "sampling/sampling_logp_difference/max": 1.8471393585205078, "sampling/sampling_logp_difference/mean": 0.02436882257461548, "step": 1069 }, { "clip_ratio/high_max": 0.006850600708276033, "clip_ratio/high_mean": 0.006850600708276033, "clip_ratio/low_mean": 0.02384992502629757, "clip_ratio/low_min": 0.02384992502629757, "clip_ratio/region_mean": 0.030700525734573603, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 36.375, "completions/mean_terminated_length": 36.375, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.26383113488554955, "epoch": 0.04297706550990079, "frac_reward_zero_std": 0.0, "grad_norm": 7.106749534606934, "learning_rate": 6.760606060606061e-06, "loss": 0.0141, "num_tokens": 2404519.0, "reward": 0.9458709359169006, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9458709359169006, "reward_meter_std": 0.06306823343038559, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0630682110786438, "reward_total_composite_mean": 0.9458709359169006, "reward_total_composite_std": 0.06306823343038559, "reward_total_mean": 0.9458709359169006, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9458709359169006, "rewards/meter/std": 0.06306823343038559, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9458709359169006, "rewards/total_composite/std": 0.06306823343038559, "sampling/importance_sampling_ratio/max": 1.6269347667694092, "sampling/importance_sampling_ratio/mean": 1.0076913833618164, "sampling/importance_sampling_ratio/min": 0.28195980191230774, "sampling/sampling_logp_difference/max": 1.2659907341003418, "sampling/sampling_logp_difference/mean": 0.043692056089639664, "step": 1070 }, { "clip_ratio/high_max": 0.0009578543831594288, "clip_ratio/high_mean": 0.0009578543831594288, "clip_ratio/low_mean": 0.006310773896984756, "clip_ratio/low_min": 0.006310773896984756, "clip_ratio/region_mean": 0.007268628280144185, "completions/clipped_ratio": 0.0, "completions/max_length": 262.0, "completions/max_terminated_length": 262.0, "completions/mean_length": 257.0, "completions/mean_terminated_length": 257.0, "completions/min_length": 249.0, "completions/min_terminated_length": 249.0, "entropy": 0.04622745281085372, "epoch": 0.043017230991685744, "frac_reward_zero_std": 0.0, "grad_norm": 1.8987226486206055, "learning_rate": 6.757575757575758e-06, "loss": 0.0028, "num_tokens": 2408287.0, "reward": 0.4727606773376465, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8534860610961914, "reward_meter_std": 0.08555062115192413, "reward_repeat_penalty_mean": 0.6499999761581421, "reward_repeat_penalty_std": 0.05909368395805359, "reward_std": 0.03482651710510254, "reward_total_composite_mean": 0.4727606773376465, "reward_total_composite_std": 0.03482650965452194, "reward_total_mean": 0.4727606773376465, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8534860610961914, "rewards/meter/std": 0.08555062115192413, "rewards/repeat_penalty/mean": 0.6499999761581421, "rewards/repeat_penalty/std": 0.05909368395805359, "rewards/total_composite/mean": 0.4727606773376465, "rewards/total_composite/std": 0.03482650965452194, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001491665840149, "sampling/importance_sampling_ratio/min": 0.29330602288246155, "sampling/sampling_logp_difference/max": 1.9065320491790771, "sampling/sampling_logp_difference/mean": 0.011769634671509266, "step": 1071 }, { "clip_ratio/high_max": 0.01675201370380819, "clip_ratio/high_mean": 0.01675201370380819, "clip_ratio/low_mean": 0.003919956274330616, "clip_ratio/low_min": 0.003919956274330616, "clip_ratio/region_mean": 0.020671969978138804, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 96.625, "completions/mean_terminated_length": 96.625, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.15043141692876816, "epoch": 0.0430573964734707, "frac_reward_zero_std": 0.0, "grad_norm": 4.010406494140625, "learning_rate": 6.754545454545455e-06, "loss": -0.009, "num_tokens": 2410412.0, "reward": 0.6105661392211914, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7025290727615356, "reward_meter_std": 0.24330702424049377, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.21741104125976562, "reward_total_composite_mean": 0.6105661392211914, "reward_total_composite_std": 0.21741104125976562, "reward_total_mean": 0.6105661392211914, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7025290727615356, "rewards/meter/std": 0.24330702424049377, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6105661392211914, "rewards/total_composite/std": 0.21741104125976562, "sampling/importance_sampling_ratio/max": 1.5932246446609497, "sampling/importance_sampling_ratio/mean": 1.0027921199798584, "sampling/importance_sampling_ratio/min": 0.3497931957244873, "sampling/sampling_logp_difference/max": 1.0504131317138672, "sampling/sampling_logp_difference/mean": 0.02641892619431019, "step": 1072 }, { "clip_ratio/high_max": 0.03231395175680518, "clip_ratio/high_mean": 0.03231395175680518, "clip_ratio/low_mean": 0.012625776929780841, "clip_ratio/low_min": 0.012625776929780841, "clip_ratio/region_mean": 0.04493972868658602, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 122.375, "completions/mean_terminated_length": 122.375, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.3304913900792599, "epoch": 0.04309756195525565, "frac_reward_zero_std": 0.0, "grad_norm": 4.956683158874512, "learning_rate": 6.751515151515152e-06, "loss": -0.0119, "num_tokens": 2412879.0, "reward": 0.8155930042266846, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8654184341430664, "reward_meter_std": 0.3119131624698639, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.30145007371902466, "reward_total_composite_mean": 0.8155930042266846, "reward_total_composite_std": 0.30145007371902466, "reward_total_mean": 0.8155930042266846, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8654184341430664, "rewards/meter/std": 0.3119131624698639, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8155930042266846, "rewards/total_composite/std": 0.30145007371902466, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0030038356781006, "sampling/importance_sampling_ratio/min": 0.2730752229690552, "sampling/sampling_logp_difference/max": 1.2980079650878906, "sampling/sampling_logp_difference/mean": 0.04843366891145706, "step": 1073 }, { "clip_ratio/high_max": 0.0071886846562847495, "clip_ratio/high_mean": 0.0071886846562847495, "clip_ratio/low_mean": 0.0050951789889950305, "clip_ratio/low_min": 0.0050951789889950305, "clip_ratio/region_mean": 0.01228386364527978, "completions/clipped_ratio": 0.0, "completions/max_length": 289.0, "completions/max_terminated_length": 289.0, "completions/mean_length": 276.25, "completions/mean_terminated_length": 276.25, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "entropy": 0.1006718729622662, "epoch": 0.043137727437040606, "frac_reward_zero_std": 0.0, "grad_norm": 2.035898208618164, "learning_rate": 6.748484848484848e-06, "loss": 0.0024, "num_tokens": 2416593.0, "reward": 0.5867777466773987, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625821024179459, "reward_meter_mean": 0.9123198986053467, "reward_meter_std": 0.15345944464206696, "reward_repeat_penalty_mean": 0.6886509656906128, "reward_repeat_penalty_std": 0.08202779293060303, "reward_std": 0.12737217545509338, "reward_total_composite_mean": 0.5867777466773987, "reward_total_composite_std": 0.12737217545509338, "reward_total_mean": 0.5867777466773987, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625821024179459, "rewards/meter/mean": 0.9123198986053467, "rewards/meter/std": 0.15345944464206696, "rewards/repeat_penalty/mean": 0.6886509656906128, "rewards/repeat_penalty/std": 0.08202779293060303, "rewards/total_composite/mean": 0.5867777466773987, "rewards/total_composite/std": 0.12737217545509338, "sampling/importance_sampling_ratio/max": 1.5070128440856934, "sampling/importance_sampling_ratio/mean": 0.9994420409202576, "sampling/importance_sampling_ratio/min": 0.21379627287387848, "sampling/sampling_logp_difference/max": 1.542731761932373, "sampling/sampling_logp_difference/mean": 0.01869548298418522, "step": 1074 }, { "clip_ratio/high_max": 0.001742160296998918, "clip_ratio/high_mean": 0.001742160296998918, "clip_ratio/low_mean": 0.0065517745388206095, "clip_ratio/low_min": 0.0065517745388206095, "clip_ratio/region_mean": 0.008293934835819528, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 290.75, "completions/mean_terminated_length": 290.75, "completions/min_length": 268.0, "completions/min_terminated_length": 268.0, "entropy": 0.08378143096342683, "epoch": 0.04317789291882556, "frac_reward_zero_std": 0.0, "grad_norm": 1.7712578773498535, "learning_rate": 6.7454545454545465e-06, "loss": 0.012, "num_tokens": 2420319.0, "reward": 0.6113426089286804, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9649861454963684, "reward_meter_std": 0.05583993345499039, "reward_repeat_penalty_mean": 0.634615421295166, "reward_repeat_penalty_std": 0.03560846298933029, "reward_std": 0.032762277871370316, "reward_total_composite_mean": 0.6113426089286804, "reward_total_composite_std": 0.03276226669549942, "reward_total_mean": 0.6113426089286804, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9649861454963684, "rewards/meter/std": 0.05583993345499039, "rewards/repeat_penalty/mean": 0.634615421295166, "rewards/repeat_penalty/std": 0.03560846298933029, "rewards/total_composite/mean": 0.6113426089286804, "rewards/total_composite/std": 0.03276226669549942, "sampling/importance_sampling_ratio/max": 1.738738775253296, "sampling/importance_sampling_ratio/mean": 0.9999510049819946, "sampling/importance_sampling_ratio/min": 0.2570464313030243, "sampling/sampling_logp_difference/max": 1.3584985733032227, "sampling/sampling_logp_difference/mean": 0.012581315822899342, "step": 1075 }, { "clip_ratio/high_max": 0.013822466367855668, "clip_ratio/high_mean": 0.013822466367855668, "clip_ratio/low_mean": 0.012700201012194157, "clip_ratio/low_min": 0.012700201012194157, "clip_ratio/region_mean": 0.026522667380049825, "completions/clipped_ratio": 0.0, "completions/max_length": 186.0, "completions/max_terminated_length": 186.0, "completions/mean_length": 170.0, "completions/mean_terminated_length": 170.0, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.14643115364015102, "epoch": 0.043218058400610514, "frac_reward_zero_std": 0.0, "grad_norm": 4.413632869720459, "learning_rate": 6.742424242424243e-06, "loss": -0.0057, "num_tokens": 2423119.0, "reward": 0.4233958125114441, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7671191692352295, "reward_meter_std": 0.2540895640850067, "reward_repeat_penalty_mean": 0.7361111044883728, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.14756235480308533, "reward_total_composite_mean": 0.4233958125114441, "reward_total_composite_std": 0.14756236970424652, "reward_total_mean": 0.4233958125114441, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7671191692352295, "rewards/meter/std": 0.2540895640850067, "rewards/repeat_penalty/mean": 0.7361111044883728, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.4233958125114441, "rewards/total_composite/std": 0.14756236970424652, "sampling/importance_sampling_ratio/max": 1.7489451169967651, "sampling/importance_sampling_ratio/mean": 0.993308961391449, "sampling/importance_sampling_ratio/min": 0.0001487771951360628, "sampling/sampling_logp_difference/max": 8.813060760498047, "sampling/sampling_logp_difference/mean": 0.04489843547344208, "step": 1076 }, { "clip_ratio/high_max": 0.0065830720122903585, "clip_ratio/high_mean": 0.0065830720122903585, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/region_mean": 0.00870171608403325, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 57.5, "completions/mean_terminated_length": 57.5, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.07559043914079666, "epoch": 0.04325822388239547, "frac_reward_zero_std": 0.0, "grad_norm": 3.8939266204833984, "learning_rate": 6.73939393939394e-06, "loss": 0.012, "num_tokens": 2424867.0, "reward": 0.6582721471786499, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9874081611633301, "reward_meter_std": 0.004490252584218979, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002993511501699686, "reward_total_composite_mean": 0.6582721471786499, "reward_total_composite_std": 0.0029935124330222607, "reward_total_mean": 0.6582721471786499, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9874081611633301, "rewards/meter/std": 0.004490252584218979, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6582721471786499, "rewards/total_composite/std": 0.0029935124330222607, "sampling/importance_sampling_ratio/max": 1.406072735786438, "sampling/importance_sampling_ratio/mean": 1.0056201219558716, "sampling/importance_sampling_ratio/min": 0.7047555446624756, "sampling/sampling_logp_difference/max": 0.34990429878234863, "sampling/sampling_logp_difference/mean": 0.010735772550106049, "step": 1077 }, { "clip_ratio/high_max": 0.0006038647261448205, "clip_ratio/high_mean": 0.0006038647261448205, "clip_ratio/low_mean": 0.002776339795673266, "clip_ratio/low_min": 0.002776339795673266, "clip_ratio/region_mean": 0.0033802045218180865, "completions/clipped_ratio": 0.0, "completions/max_length": 433.0, "completions/max_terminated_length": 433.0, "completions/mean_length": 403.25, "completions/mean_terminated_length": 403.25, "completions/min_length": 384.0, "completions/min_terminated_length": 384.0, "entropy": 0.03162188641726971, "epoch": 0.04329838936418042, "frac_reward_zero_std": 0.0, "grad_norm": 0.8265510201454163, "learning_rate": 6.7363636363636365e-06, "loss": -0.0117, "num_tokens": 2429621.0, "reward": 0.38671886920928955, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9796521067619324, "reward_meter_std": 0.0062021855264902115, "reward_repeat_penalty_mean": 0.5921052694320679, "reward_repeat_penalty_std": 0.037216152995824814, "reward_std": 0.024711309000849724, "reward_total_composite_mean": 0.38671886920928955, "reward_total_composite_std": 0.02471131458878517, "reward_total_mean": 0.38671886920928955, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9796521067619324, "rewards/meter/std": 0.0062021855264902115, "rewards/repeat_penalty/mean": 0.5921052694320679, "rewards/repeat_penalty/std": 0.037216152995824814, "rewards/total_composite/mean": 0.38671886920928955, "rewards/total_composite/std": 0.02471131458878517, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0002822875976562, "sampling/importance_sampling_ratio/min": 0.2117329090833664, "sampling/sampling_logp_difference/max": 1.5524296760559082, "sampling/sampling_logp_difference/mean": 0.005612175911664963, "step": 1078 }, { "clip_ratio/high_max": 0.005959982983767986, "clip_ratio/high_mean": 0.005959982983767986, "clip_ratio/low_mean": 0.0014367816038429737, "clip_ratio/low_min": 0.0014367816038429737, "clip_ratio/region_mean": 0.00739676458761096, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 84.0, "completions/mean_terminated_length": 84.0, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.04214879055507481, "epoch": 0.043338554845965375, "frac_reward_zero_std": 0.0, "grad_norm": 4.684656143188477, "learning_rate": 6.733333333333334e-06, "loss": 0.0067, "num_tokens": 2431573.0, "reward": 0.9432131052017212, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9432131052017212, "reward_meter_std": 0.005848580971360207, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0058485823683440685, "reward_total_composite_mean": 0.9432131052017212, "reward_total_composite_std": 0.005848580971360207, "reward_total_mean": 0.9432131052017212, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9432131052017212, "rewards/meter/std": 0.005848580971360207, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9432131052017212, "rewards/total_composite/std": 0.005848580971360207, "sampling/importance_sampling_ratio/max": 1.180408239364624, "sampling/importance_sampling_ratio/mean": 0.9998735785484314, "sampling/importance_sampling_ratio/min": 0.2840469479560852, "sampling/sampling_logp_difference/max": 1.2586157321929932, "sampling/sampling_logp_difference/mean": 0.00942947156727314, "step": 1079 }, { "clip_ratio/high_max": 0.014718614984303713, "clip_ratio/high_mean": 0.014718614984303713, "clip_ratio/low_mean": 0.022774446289986372, "clip_ratio/low_min": 0.022774446289986372, "clip_ratio/region_mean": 0.037493061274290085, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 57.75, "completions/mean_terminated_length": 57.75, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.14285273849964142, "epoch": 0.04337872032775033, "frac_reward_zero_std": 0.0, "grad_norm": 7.663031578063965, "learning_rate": 6.73030303030303e-06, "loss": -0.088, "num_tokens": 2433443.0, "reward": 0.31589365005493164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.31589365005493164, "reward_meter_std": 0.41653454303741455, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.41653454303741455, "reward_total_composite_mean": 0.31589365005493164, "reward_total_composite_std": 0.41653454303741455, "reward_total_mean": 0.31589365005493164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.31589365005493164, "rewards/meter/std": 0.41653454303741455, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.31589365005493164, "rewards/total_composite/std": 0.41653454303741455, "sampling/importance_sampling_ratio/max": 1.726487159729004, "sampling/importance_sampling_ratio/mean": 1.0007740259170532, "sampling/importance_sampling_ratio/min": 0.22385750710964203, "sampling/sampling_logp_difference/max": 1.4967455863952637, "sampling/sampling_logp_difference/mean": 0.038866013288497925, "step": 1080 }, { "clip_ratio/high_max": 0.0029411765281111, "clip_ratio/high_mean": 0.0029411765281111, "clip_ratio/low_mean": 0.010382481734268367, "clip_ratio/low_min": 0.010382481734268367, "clip_ratio/region_mean": 0.013323658262379467, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 84.25, "completions/mean_terminated_length": 84.25, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.06444978155195713, "epoch": 0.04341888580953528, "frac_reward_zero_std": 0.0, "grad_norm": 2.3848185539245605, "learning_rate": 6.7272727272727275e-06, "loss": 0.0001, "num_tokens": 2435373.0, "reward": 0.8173696994781494, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9907380938529968, "reward_meter_std": 0.0005972905782982707, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07022644579410553, "reward_total_composite_mean": 0.8173696994781494, "reward_total_composite_std": 0.07022644579410553, "reward_total_mean": 0.8173696994781494, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9907380938529968, "rewards/meter/std": 0.0005972905782982707, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8173696994781494, "rewards/total_composite/std": 0.07022644579410553, "sampling/importance_sampling_ratio/max": 1.8647558689117432, "sampling/importance_sampling_ratio/mean": 0.9998801946640015, "sampling/importance_sampling_ratio/min": 0.40769892930984497, "sampling/sampling_logp_difference/max": 0.8972263336181641, "sampling/sampling_logp_difference/mean": 0.018476024270057678, "step": 1081 }, { "clip_ratio/high_max": 0.009705353993922472, "clip_ratio/high_mean": 0.009705353993922472, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/region_mean": 0.011441465117968619, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 77.625, "completions/mean_terminated_length": 77.625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.11937720235437155, "epoch": 0.04345905129132024, "frac_reward_zero_std": 0.0, "grad_norm": 3.134387969970703, "learning_rate": 6.724242424242424e-06, "loss": -0.0216, "num_tokens": 2437378.0, "reward": 0.9952170252799988, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952170252799988, "reward_meter_std": 0.005060417577624321, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005060393828898668, "reward_total_composite_mean": 0.9952170252799988, "reward_total_composite_std": 0.005060417577624321, "reward_total_mean": 0.9952170252799988, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952170252799988, "rewards/meter/std": 0.005060417577624321, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952170252799988, "rewards/total_composite/std": 0.005060417577624321, "sampling/importance_sampling_ratio/max": 1.727461814880371, "sampling/importance_sampling_ratio/mean": 1.0053282976150513, "sampling/importance_sampling_ratio/min": 0.17540933191776276, "sampling/sampling_logp_difference/max": 1.7406330108642578, "sampling/sampling_logp_difference/mean": 0.023099998012185097, "step": 1082 }, { "clip_ratio/high_max": 0.01804604707285762, "clip_ratio/high_mean": 0.01804604707285762, "clip_ratio/low_mean": 0.0030412437627092004, "clip_ratio/low_min": 0.0030412437627092004, "clip_ratio/region_mean": 0.02108729083556682, "completions/clipped_ratio": 0.0, "completions/max_length": 135.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 127.75, "completions/mean_terminated_length": 127.75, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.10621183086186647, "epoch": 0.04349921677310519, "frac_reward_zero_std": 0.0, "grad_norm": 5.954423427581787, "learning_rate": 6.721212121212122e-06, "loss": -0.0062, "num_tokens": 2439776.0, "reward": 0.43424269556999207, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7971546053886414, "reward_meter_std": 0.2483818531036377, "reward_repeat_penalty_mean": 0.7361111640930176, "reward_repeat_penalty_std": 0.13197055459022522, "reward_std": 0.14866070449352264, "reward_total_composite_mean": 0.43424269556999207, "reward_total_composite_std": 0.14866070449352264, "reward_total_mean": 0.43424269556999207, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7971546053886414, "rewards/meter/std": 0.2483818531036377, "rewards/repeat_penalty/mean": 0.7361111640930176, "rewards/repeat_penalty/std": 0.13197055459022522, "rewards/total_composite/mean": 0.43424269556999207, "rewards/total_composite/std": 0.14866070449352264, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028880834579468, "sampling/importance_sampling_ratio/min": 0.14491204917430878, "sampling/sampling_logp_difference/max": 1.9316282272338867, "sampling/sampling_logp_difference/mean": 0.02712804079055786, "step": 1083 }, { "clip_ratio/high_max": 0.018721832893788815, "clip_ratio/high_mean": 0.018721832893788815, "clip_ratio/low_mean": 0.009149131015874445, "clip_ratio/low_min": 0.009149131015874445, "clip_ratio/region_mean": 0.02787096390966326, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 80.875, "completions/mean_terminated_length": 80.875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.10348887462168932, "epoch": 0.043539382254890145, "frac_reward_zero_std": 0.0, "grad_norm": 7.175000190734863, "learning_rate": 6.718181818181819e-06, "loss": 0.005, "num_tokens": 2441759.0, "reward": 0.9879165291786194, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9879165291786194, "reward_meter_std": 0.011460079811513424, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011460077948868275, "reward_total_composite_mean": 0.9879165291786194, "reward_total_composite_std": 0.011460079811513424, "reward_total_mean": 0.9879165291786194, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9879165291786194, "rewards/meter/std": 0.011460079811513424, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9879165291786194, "rewards/total_composite/std": 0.011460079811513424, "sampling/importance_sampling_ratio/max": 1.9328581094741821, "sampling/importance_sampling_ratio/mean": 0.9987949132919312, "sampling/importance_sampling_ratio/min": 0.20207108557224274, "sampling/sampling_logp_difference/max": 1.5991357564926147, "sampling/sampling_logp_difference/mean": 0.02571716159582138, "step": 1084 }, { "clip_ratio/high_max": 0.004307354771299288, "clip_ratio/high_mean": 0.004307354771299288, "clip_ratio/low_mean": 0.0023487245198339224, "clip_ratio/low_min": 0.0023487245198339224, "clip_ratio/region_mean": 0.00665607929113321, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 335.625, "completions/mean_terminated_length": 335.625, "completions/min_length": 318.0, "completions/min_terminated_length": 318.0, "entropy": 0.034420924610458314, "epoch": 0.0435795477366751, "frac_reward_zero_std": 0.0, "grad_norm": 0.6841692924499512, "learning_rate": 6.715151515151516e-06, "loss": -0.03, "num_tokens": 2446044.0, "reward": 0.5644394755363464, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.05750546231865883, "reward_meter_mean": 0.9889903664588928, "reward_meter_std": 0.017724184319376945, "reward_repeat_penalty_mean": 0.5962929129600525, "reward_repeat_penalty_std": 0.030314533039927483, "reward_std": 0.03291511535644531, "reward_total_composite_mean": 0.5644394755363464, "reward_total_composite_std": 0.03291511908173561, "reward_total_mean": 0.5644394755363464, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.05750546231865883, "rewards/meter/mean": 0.9889903664588928, "rewards/meter/std": 0.017724184319376945, "rewards/repeat_penalty/mean": 0.5962929129600525, "rewards/repeat_penalty/std": 0.030314533039927483, "rewards/total_composite/mean": 0.5644394755363464, "rewards/total_composite/std": 0.03291511908173561, "sampling/importance_sampling_ratio/max": 1.9612756967544556, "sampling/importance_sampling_ratio/mean": 0.9996162056922913, "sampling/importance_sampling_ratio/min": 0.3504544794559479, "sampling/sampling_logp_difference/max": 1.0485243797302246, "sampling/sampling_logp_difference/mean": 0.008025525137782097, "step": 1085 }, { "clip_ratio/high_max": 0.0027567246870603412, "clip_ratio/high_mean": 0.0027567246870603412, "clip_ratio/low_mean": 0.0010880097979679704, "clip_ratio/low_min": 0.0010880097979679704, "clip_ratio/region_mean": 0.0038447344850283116, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 350.5, "completions/mean_terminated_length": 350.5, "completions/min_length": 344.0, "completions/min_terminated_length": 344.0, "entropy": 0.025596523424610496, "epoch": 0.04361971321846005, "frac_reward_zero_std": 0.0, "grad_norm": 0.8804117441177368, "learning_rate": 6.712121212121213e-06, "loss": -0.0088, "num_tokens": 2450552.0, "reward": 0.5277966260910034, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969491958618164, "reward_meter_std": 0.0016905681695789099, "reward_repeat_penalty_mean": 0.5882353186607361, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00089502043556422, "reward_total_composite_mean": 0.5277966260910034, "reward_total_composite_std": 0.0008950114133767784, "reward_total_mean": 0.5277966260910034, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969491958618164, "rewards/meter/std": 0.0016905681695789099, "rewards/repeat_penalty/mean": 0.5882353186607361, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5277966260910034, "rewards/total_composite/std": 0.0008950114133767784, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.999923050403595, "sampling/importance_sampling_ratio/min": 0.3424888849258423, "sampling/sampling_logp_difference/max": 1.0715160369873047, "sampling/sampling_logp_difference/mean": 0.005859900265932083, "step": 1086 }, { "clip_ratio/high_max": 0.01255706837400794, "clip_ratio/high_mean": 0.01255706837400794, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/region_mean": 0.01931382529437542, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 79.375, "completions/mean_terminated_length": 79.375, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.2076272815465927, "epoch": 0.04365987870024501, "frac_reward_zero_std": 0.0, "grad_norm": 3.9621143341064453, "learning_rate": 6.709090909090909e-06, "loss": -0.0062, "num_tokens": 2452523.0, "reward": 0.7115691900253296, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7115691900253296, "reward_meter_std": 0.38112178444862366, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.38112178444862366, "reward_total_composite_mean": 0.7115691900253296, "reward_total_composite_std": 0.38112178444862366, "reward_total_mean": 0.7115691900253296, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7115691900253296, "rewards/meter/std": 0.38112178444862366, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7115691900253296, "rewards/total_composite/std": 0.38112178444862366, "sampling/importance_sampling_ratio/max": 1.846152663230896, "sampling/importance_sampling_ratio/mean": 1.0051707029342651, "sampling/importance_sampling_ratio/min": 0.1325182169675827, "sampling/sampling_logp_difference/max": 2.0210351943969727, "sampling/sampling_logp_difference/mean": 0.034753963351249695, "step": 1087 }, { "clip_ratio/high_max": 0.04101851023733616, "clip_ratio/high_mean": 0.04101851023733616, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/region_mean": 0.05238214693963528, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 33.5, "completions/mean_terminated_length": 33.5, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.18665132857859135, "epoch": 0.04370004418202996, "frac_reward_zero_std": 0.0, "grad_norm": 6.578573226928711, "learning_rate": 6.706060606060607e-06, "loss": -0.0029, "num_tokens": 2454071.0, "reward": 0.7622437477111816, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7622437477111816, "reward_meter_std": 0.3664873540401459, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3664873540401459, "reward_total_composite_mean": 0.7622437477111816, "reward_total_composite_std": 0.3664873540401459, "reward_total_mean": 0.7622437477111816, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7622437477111816, "rewards/meter/std": 0.3664873540401459, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7622437477111816, "rewards/total_composite/std": 0.3664873540401459, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0093433856964111, "sampling/importance_sampling_ratio/min": 0.3433097004890442, "sampling/sampling_logp_difference/max": 1.069122314453125, "sampling/sampling_logp_difference/mean": 0.040684185922145844, "step": 1088 }, { "clip_ratio/high_max": 0.03449019626714289, "clip_ratio/high_mean": 0.03449019626714289, "clip_ratio/low_mean": 0.009895833441987634, "clip_ratio/low_min": 0.009895833441987634, "clip_ratio/region_mean": 0.044386029709130526, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 58.625, "completions/mean_terminated_length": 58.625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.1829883987084031, "epoch": 0.043740209663814915, "frac_reward_zero_std": 0.0, "grad_norm": 4.875842571258545, "learning_rate": 6.703030303030304e-06, "loss": 0.0411, "num_tokens": 2455804.0, "reward": 0.8143894672393799, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8143894672393799, "reward_meter_std": 0.22199998795986176, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22199997305870056, "reward_total_composite_mean": 0.8143894672393799, "reward_total_composite_std": 0.22199998795986176, "reward_total_mean": 0.8143894672393799, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8143894672393799, "rewards/meter/std": 0.22199998795986176, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8143894672393799, "rewards/total_composite/std": 0.22199998795986176, "sampling/importance_sampling_ratio/max": 1.836875081062317, "sampling/importance_sampling_ratio/mean": 0.9986065030097961, "sampling/importance_sampling_ratio/min": 0.19256533682346344, "sampling/sampling_logp_difference/max": 1.6473197937011719, "sampling/sampling_logp_difference/mean": 0.04701100289821625, "step": 1089 }, { "clip_ratio/high_max": 0.012134980701375753, "clip_ratio/high_mean": 0.012134980701375753, "clip_ratio/low_mean": 0.004208970727631822, "clip_ratio/low_min": 0.004208970727631822, "clip_ratio/region_mean": 0.016343951429007575, "completions/clipped_ratio": 0.0, "completions/max_length": 288.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 271.25, "completions/mean_terminated_length": 271.25, "completions/min_length": 258.0, "completions/min_terminated_length": 258.0, "entropy": 0.10771761322394013, "epoch": 0.04378037514559987, "frac_reward_zero_std": 0.0, "grad_norm": 1.876051902770996, "learning_rate": 6.700000000000001e-06, "loss": -0.0213, "num_tokens": 2459750.0, "reward": 0.5263446569442749, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9913715124130249, "reward_meter_std": 0.014886174350976944, "reward_repeat_penalty_mean": 0.5326797366142273, "reward_repeat_penalty_std": 0.2576701045036316, "reward_std": 0.2508489787578583, "reward_total_composite_mean": 0.5263446569442749, "reward_total_composite_std": 0.2508489787578583, "reward_total_mean": 0.5263446569442749, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9913715124130249, "rewards/meter/std": 0.014886174350976944, "rewards/repeat_penalty/mean": 0.5326797366142273, "rewards/repeat_penalty/std": 0.2576701045036316, "rewards/total_composite/mean": 0.5263446569442749, "rewards/total_composite/std": 0.2508489787578583, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0003299713134766, "sampling/importance_sampling_ratio/min": 0.275922030210495, "sampling/sampling_logp_difference/max": 1.2876369953155518, "sampling/sampling_logp_difference/mean": 0.019940270110964775, "step": 1090 }, { "clip_ratio/high_max": 0.005542142200283706, "clip_ratio/high_mean": 0.005542142200283706, "clip_ratio/low_mean": 0.021997954230755568, "clip_ratio/low_min": 0.021997954230755568, "clip_ratio/region_mean": 0.027540096431039274, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 64.75, "completions/mean_terminated_length": 64.75, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.27551514096558094, "epoch": 0.04382054062738482, "frac_reward_zero_std": 0.0, "grad_norm": 6.454976558685303, "learning_rate": 6.6969696969696975e-06, "loss": -0.03, "num_tokens": 2461500.0, "reward": 0.5685036182403564, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5685036182403564, "reward_meter_std": 0.32084769010543823, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.32084769010543823, "reward_total_composite_mean": 0.5685036182403564, "reward_total_composite_std": 0.32084769010543823, "reward_total_mean": 0.5685036182403564, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5685036182403564, "rewards/meter/std": 0.32084769010543823, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5685036182403564, "rewards/total_composite/std": 0.32084769010543823, "sampling/importance_sampling_ratio/max": 1.7851393222808838, "sampling/importance_sampling_ratio/mean": 1.007980227470398, "sampling/importance_sampling_ratio/min": 0.2333618849515915, "sampling/sampling_logp_difference/max": 1.455164909362793, "sampling/sampling_logp_difference/mean": 0.046030234545469284, "step": 1091 }, { "clip_ratio/high_max": 0.006234930478967726, "clip_ratio/high_mean": 0.006234930478967726, "clip_ratio/low_mean": 0.0030868902103975415, "clip_ratio/low_min": 0.0030868902103975415, "clip_ratio/region_mean": 0.009321820689365268, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 79.625, "completions/mean_terminated_length": 79.625, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.07495173905044794, "epoch": 0.04386070610916978, "frac_reward_zero_std": 0.0, "grad_norm": 4.461249351501465, "learning_rate": 6.693939393939395e-06, "loss": 0.0095, "num_tokens": 2463457.0, "reward": 0.8799495697021484, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8799495697021484, "reward_meter_std": 0.1550467163324356, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1550467312335968, "reward_total_composite_mean": 0.8799495697021484, "reward_total_composite_std": 0.1550467163324356, "reward_total_mean": 0.8799495697021484, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8799495697021484, "rewards/meter/std": 0.1550467163324356, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8799495697021484, "rewards/total_composite/std": 0.1550467163324356, "sampling/importance_sampling_ratio/max": 1.76300048828125, "sampling/importance_sampling_ratio/mean": 1.0008015632629395, "sampling/importance_sampling_ratio/min": 0.09676557034254074, "sampling/sampling_logp_difference/max": 2.3354640007019043, "sampling/sampling_logp_difference/mean": 0.021693911403417587, "step": 1092 }, { "clip_ratio/high_max": 0.021643388201482594, "clip_ratio/high_mean": 0.021643388201482594, "clip_ratio/low_mean": 0.012298387009650469, "clip_ratio/low_min": 0.012298387009650469, "clip_ratio/region_mean": 0.03394177521113306, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 62.5, "completions/mean_terminated_length": 62.5, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.1802924033254385, "epoch": 0.04390087159095473, "frac_reward_zero_std": 0.0, "grad_norm": 10.553791046142578, "learning_rate": 6.690909090909091e-06, "loss": 0.0004, "num_tokens": 2465181.0, "reward": 0.8936691284179688, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8936691284179688, "reward_meter_std": 0.16734978556632996, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.16734978556632996, "reward_total_composite_mean": 0.8936691284179688, "reward_total_composite_std": 0.16734978556632996, "reward_total_mean": 0.8936691284179688, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8936691284179688, "rewards/meter/std": 0.16734978556632996, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8936691284179688, "rewards/total_composite/std": 0.16734978556632996, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0026055574417114, "sampling/importance_sampling_ratio/min": 0.20930436253547668, "sampling/sampling_logp_difference/max": 1.5639657974243164, "sampling/sampling_logp_difference/mean": 0.03492492809891701, "step": 1093 }, { "clip_ratio/high_max": 0.011392419459298253, "clip_ratio/high_mean": 0.011392419459298253, "clip_ratio/low_mean": 0.0017985611921176314, "clip_ratio/low_min": 0.0017985611921176314, "clip_ratio/region_mean": 0.013190980651415884, "completions/clipped_ratio": 0.0, "completions/max_length": 139.0, "completions/max_terminated_length": 139.0, "completions/mean_length": 136.0, "completions/mean_terminated_length": 136.0, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.07300170278176665, "epoch": 0.043941037072739685, "frac_reward_zero_std": 0.0, "grad_norm": 2.7043662071228027, "learning_rate": 6.687878787878788e-06, "loss": 0.0321, "num_tokens": 2467653.0, "reward": 0.41557225584983826, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7940855622291565, "reward_meter_std": 0.11631763726472855, "reward_repeat_penalty_mean": 0.6944444179534912, "reward_repeat_penalty_std": 0.05143444612622261, "reward_std": 0.08242405205965042, "reward_total_composite_mean": 0.41557225584983826, "reward_total_composite_std": 0.08242405205965042, "reward_total_mean": 0.41557225584983826, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7940855622291565, "rewards/meter/std": 0.11631763726472855, "rewards/repeat_penalty/mean": 0.6944444179534912, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.41557225584983826, "rewards/total_composite/std": 0.08242405205965042, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9989469051361084, "sampling/importance_sampling_ratio/min": 0.2393883466720581, "sampling/sampling_logp_difference/max": 1.4296681880950928, "sampling/sampling_logp_difference/mean": 0.01717483066022396, "step": 1094 }, { "clip_ratio/high_max": 0.031376007944345474, "clip_ratio/high_mean": 0.031376007944345474, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/region_mean": 0.03970934171229601, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 31.75, "completions/mean_terminated_length": 31.75, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.16218565311282873, "epoch": 0.04398120255452464, "frac_reward_zero_std": 0.0, "grad_norm": 11.521310806274414, "learning_rate": 6.684848484848485e-06, "loss": -0.0163, "num_tokens": 2469179.0, "reward": 0.8792948722839355, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8792948722839355, "reward_meter_std": 0.25036633014678955, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.25036633014678955, "reward_total_composite_mean": 0.8792948722839355, "reward_total_composite_std": 0.25036633014678955, "reward_total_mean": 0.8792948722839355, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8792948722839355, "rewards/meter/std": 0.25036633014678955, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8792948722839355, "rewards/total_composite/std": 0.25036633014678955, "sampling/importance_sampling_ratio/max": 1.482624888420105, "sampling/importance_sampling_ratio/mean": 0.9937551617622375, "sampling/importance_sampling_ratio/min": 0.23205561935901642, "sampling/sampling_logp_difference/max": 1.4607782363891602, "sampling/sampling_logp_difference/mean": 0.04188957437872887, "step": 1095 }, { "clip_ratio/high_max": 0.019367096945643425, "clip_ratio/high_mean": 0.019367096945643425, "clip_ratio/low_mean": 0.005780933075584471, "clip_ratio/low_min": 0.005780933075584471, "clip_ratio/region_mean": 0.025148030021227896, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 81.0, "completions/mean_terminated_length": 81.0, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.15196853037923574, "epoch": 0.04402136803630959, "frac_reward_zero_std": 0.0, "grad_norm": 4.185685634613037, "learning_rate": 6.681818181818183e-06, "loss": 0.0349, "num_tokens": 2471195.0, "reward": 0.7183074951171875, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8235281705856323, "reward_meter_std": 0.1984855979681015, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.19789229333400726, "reward_total_composite_mean": 0.7183074951171875, "reward_total_composite_std": 0.19789229333400726, "reward_total_mean": 0.7183074951171875, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8235281705856323, "rewards/meter/std": 0.1984855979681015, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.7183074951171875, "rewards/total_composite/std": 0.19789229333400726, "sampling/importance_sampling_ratio/max": 1.7130012512207031, "sampling/importance_sampling_ratio/mean": 0.9967372417449951, "sampling/importance_sampling_ratio/min": 0.262839138507843, "sampling/sampling_logp_difference/max": 1.3362131118774414, "sampling/sampling_logp_difference/mean": 0.03307102620601654, "step": 1096 }, { "clip_ratio/high_max": 0.01060027233324945, "clip_ratio/high_mean": 0.01060027233324945, "clip_ratio/low_mean": 0.010694848489947617, "clip_ratio/low_min": 0.010694848489947617, "clip_ratio/region_mean": 0.021295120823197067, "completions/clipped_ratio": 0.0, "completions/max_length": 320.0, "completions/max_terminated_length": 320.0, "completions/mean_length": 304.5, "completions/mean_terminated_length": 304.5, "completions/min_length": 280.0, "completions/min_terminated_length": 280.0, "entropy": 0.16177368070930243, "epoch": 0.044061533518094546, "frac_reward_zero_std": 0.0, "grad_norm": 1.8824577331542969, "learning_rate": 6.678787878787879e-06, "loss": -0.0005, "num_tokens": 2475359.0, "reward": 0.49607303738594055, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7403846383094788, "reward_count_adherence_std": 0.039811473339796066, "reward_meter_mean": 0.953475832939148, "reward_meter_std": 0.11951109021902084, "reward_repeat_penalty_mean": 0.6966374516487122, "reward_repeat_penalty_std": 0.09851117432117462, "reward_std": 0.11844708025455475, "reward_total_composite_mean": 0.49607303738594055, "reward_total_composite_std": 0.11844708025455475, "reward_total_mean": 0.49607303738594055, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7403846383094788, "rewards/count_adherence/std": 0.039811473339796066, "rewards/meter/mean": 0.953475832939148, "rewards/meter/std": 0.11951109021902084, "rewards/repeat_penalty/mean": 0.6966374516487122, "rewards/repeat_penalty/std": 0.09851117432117462, "rewards/total_composite/mean": 0.49607303738594055, "rewards/total_composite/std": 0.11844708025455475, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0036851167678833, "sampling/importance_sampling_ratio/min": 0.05991671606898308, "sampling/sampling_logp_difference/max": 2.8147997856140137, "sampling/sampling_logp_difference/mean": 0.027330797165632248, "step": 1097 }, { "clip_ratio/high_max": 0.014381667831912637, "clip_ratio/high_mean": 0.014381667831912637, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/region_mean": 0.01636579493060708, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.625, "completions/mean_terminated_length": 61.625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.08917439868673682, "epoch": 0.0441016989998795, "frac_reward_zero_std": 0.0, "grad_norm": 6.93865966796875, "learning_rate": 6.6757575757575766e-06, "loss": 0.0173, "num_tokens": 2477124.0, "reward": 0.876738965511322, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.876738965511322, "reward_meter_std": 0.2505471706390381, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2505471408367157, "reward_total_composite_mean": 0.876738965511322, "reward_total_composite_std": 0.2505471706390381, "reward_total_mean": 0.876738965511322, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.876738965511322, "rewards/meter/std": 0.2505471706390381, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.876738965511322, "rewards/total_composite/std": 0.2505471706390381, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015573501586914, "sampling/importance_sampling_ratio/min": 0.3506268262863159, "sampling/sampling_logp_difference/max": 1.0480327606201172, "sampling/sampling_logp_difference/mean": 0.026354925706982613, "step": 1098 }, { "clip_ratio/high_max": 0.009313070855569094, "clip_ratio/high_mean": 0.009313070855569094, "clip_ratio/low_mean": 0.0041160593973472714, "clip_ratio/low_min": 0.0041160593973472714, "clip_ratio/region_mean": 0.013429130252916366, "completions/clipped_ratio": 0.0, "completions/max_length": 224.0, "completions/max_terminated_length": 224.0, "completions/mean_length": 205.25, "completions/mean_terminated_length": 205.25, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "entropy": 0.09231163933873177, "epoch": 0.044141864481664454, "frac_reward_zero_std": 0.0, "grad_norm": 3.1013591289520264, "learning_rate": 6.672727272727273e-06, "loss": -0.0507, "num_tokens": 2480198.0, "reward": 0.373015820980072, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7326644062995911, "reward_meter_std": 0.39641380310058594, "reward_repeat_penalty_mean": 0.6022727489471436, "reward_repeat_penalty_std": 0.14527180790901184, "reward_std": 0.22151786088943481, "reward_total_composite_mean": 0.373015820980072, "reward_total_composite_std": 0.22151786088943481, "reward_total_mean": 0.373015820980072, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7326644062995911, "rewards/meter/std": 0.39641380310058594, "rewards/repeat_penalty/mean": 0.6022727489471436, "rewards/repeat_penalty/std": 0.14527180790901184, "rewards/total_composite/mean": 0.373015820980072, "rewards/total_composite/std": 0.22151786088943481, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023084878921509, "sampling/importance_sampling_ratio/min": 0.02916594408452511, "sampling/sampling_logp_difference/max": 3.5347535610198975, "sampling/sampling_logp_difference/mean": 0.017836585640907288, "step": 1099 }, { "clip_ratio/high_max": 0.010039791552117094, "clip_ratio/high_mean": 0.010039791552117094, "clip_ratio/low_mean": 0.001618139911442995, "clip_ratio/low_min": 0.001618139911442995, "clip_ratio/region_mean": 0.01165793146356009, "completions/clipped_ratio": 0.0, "completions/max_length": 345.0, "completions/max_terminated_length": 345.0, "completions/mean_length": 327.125, "completions/mean_terminated_length": 327.125, "completions/min_length": 302.0, "completions/min_terminated_length": 302.0, "entropy": 0.06283333199098706, "epoch": 0.04418202996344941, "frac_reward_zero_std": 0.0, "grad_norm": 0.9629241228103638, "learning_rate": 6.66969696969697e-06, "loss": -0.0248, "num_tokens": 2484623.0, "reward": 0.386793315410614, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7291666865348816, "reward_count_adherence_std": 0.038575831800699234, "reward_meter_mean": 0.993415355682373, "reward_meter_std": 0.007653866894543171, "reward_repeat_penalty_mean": 0.5248161554336548, "reward_repeat_penalty_std": 0.21153290569782257, "reward_std": 0.16382494568824768, "reward_total_composite_mean": 0.386793315410614, "reward_total_composite_std": 0.16382494568824768, "reward_total_mean": 0.386793315410614, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7291666865348816, "rewards/count_adherence/std": 0.038575831800699234, "rewards/meter/mean": 0.993415355682373, "rewards/meter/std": 0.007653866894543171, "rewards/repeat_penalty/mean": 0.5248161554336548, "rewards/repeat_penalty/std": 0.21153290569782257, "rewards/total_composite/mean": 0.386793315410614, "rewards/total_composite/std": 0.16382494568824768, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001177430152893, "sampling/importance_sampling_ratio/min": 0.35135865211486816, "sampling/sampling_logp_difference/max": 1.045947790145874, "sampling/sampling_logp_difference/mean": 0.010836898349225521, "step": 1100 }, { "epoch": 0.04418202996344941, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 328.61538461538464, "eval_completions/max_terminated_length": 328.61538461538464, "eval_completions/mean_length": 188.08653846153845, "eval_completions/mean_terminated_length": 188.08653846153845, "eval_completions/min_length": 57.92307692307692, "eval_completions/min_terminated_length": 57.92307692307692, "eval_entropy": 0.07903132902888152, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 2484623.0, "eval_reward": 0.385673477099492, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.8649956675676199, "eval_reward_count_adherence_std": 0.12389520039925209, "eval_reward_meter_mean": 0.6326167858563937, "eval_reward_meter_std": 0.35957076343206257, "eval_reward_repeat_penalty_mean": 0.6625011425751907, "eval_reward_repeat_penalty_std": 0.24176405255611128, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.385673477099492, "eval_reward_total_composite_std": 0.3103876068041875, "eval_reward_total_mean": 0.385673477099492, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.8649956675676199, "eval_rewards/count_adherence/std": 0.12389520039925209, "eval_rewards/meter/mean": 0.6326167858563937, "eval_rewards/meter/std": 0.35957076343206257, "eval_rewards/repeat_penalty/mean": 0.6625011425751907, "eval_rewards/repeat_penalty/std": 0.24176405255611128, "eval_rewards/total_composite/mean": 0.385673477099492, "eval_rewards/total_composite/std": 0.3103876068041875, "eval_runtime": 63.6923, "eval_samples_per_second": 1.633, "eval_sampling/importance_sampling_ratio/max": 1.3779725111447847, "eval_sampling/importance_sampling_ratio/mean": 1.0015292717860296, "eval_sampling/importance_sampling_ratio/min": 0.4163260803772853, "eval_sampling/sampling_logp_difference/max": 0.9079764164411105, "eval_sampling/sampling_logp_difference/mean": 0.008682384119870571, "eval_steps_per_second": 0.204, "step": 1100 }, { "clip_ratio/high_max": 0.02208862197585404, "clip_ratio/high_mean": 0.02208862197585404, "clip_ratio/low_mean": 0.01142857177183032, "clip_ratio/low_min": 0.01142857177183032, "clip_ratio/region_mean": 0.03351719374768436, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 55.875, "completions/mean_terminated_length": 55.875, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.22917167469859123, "epoch": 0.04422219544523436, "frac_reward_zero_std": 0.0, "grad_norm": 7.680727958679199, "learning_rate": 6.666666666666667e-06, "loss": 0.0008, "num_tokens": 2486422.0, "reward": 0.8475104570388794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8475104570388794, "reward_meter_std": 0.17176933586597443, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.17176933586597443, "reward_total_composite_mean": 0.8475104570388794, "reward_total_composite_std": 0.17176933586597443, "reward_total_mean": 0.8475104570388794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8475104570388794, "rewards/meter/std": 0.17176933586597443, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8475104570388794, "rewards/total_composite/std": 0.17176933586597443, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0003966093063354, "sampling/importance_sampling_ratio/min": 0.08991573750972748, "sampling/sampling_logp_difference/max": 2.4088823795318604, "sampling/sampling_logp_difference/mean": 0.05723501741886139, "step": 1101 }, { "clip_ratio/high_max": 0.007919449737528339, "clip_ratio/high_mean": 0.007919449737528339, "clip_ratio/low_mean": 0.004965955973602831, "clip_ratio/low_min": 0.004965955973602831, "clip_ratio/region_mean": 0.01288540571113117, "completions/clipped_ratio": 0.0, "completions/max_length": 335.0, "completions/max_terminated_length": 335.0, "completions/mean_length": 296.125, "completions/mean_terminated_length": 296.125, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.054934965912252665, "epoch": 0.04426236092701932, "frac_reward_zero_std": 0.0, "grad_norm": 9.668932914733887, "learning_rate": 6.663636363636365e-06, "loss": -0.0479, "num_tokens": 2490583.0, "reward": 0.37116703391075134, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_meter_mean": 0.7127225399017334, "reward_meter_std": 0.37384799122810364, "reward_repeat_penalty_mean": 0.5183823704719543, "reward_repeat_penalty_std": 0.17379139363765717, "reward_std": 0.24903464317321777, "reward_total_composite_mean": 0.37116703391075134, "reward_total_composite_std": 0.24903467297554016, "reward_total_mean": 0.37116703391075134, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/meter/mean": 0.7127225399017334, "rewards/meter/std": 0.37384799122810364, "rewards/repeat_penalty/mean": 0.5183823704719543, "rewards/repeat_penalty/std": 0.17379139363765717, "rewards/total_composite/mean": 0.37116703391075134, "rewards/total_composite/std": 0.24903467297554016, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0006206035614014, "sampling/importance_sampling_ratio/min": 0.016256263479590416, "sampling/sampling_logp_difference/max": 4.119277000427246, "sampling/sampling_logp_difference/mean": 0.014599828980863094, "step": 1102 }, { "clip_ratio/high_max": 0.012646116432733834, "clip_ratio/high_mean": 0.012646116432733834, "clip_ratio/low_mean": 0.005657031899318099, "clip_ratio/low_min": 0.005657031899318099, "clip_ratio/region_mean": 0.018303148332051933, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.10305365733802319, "epoch": 0.04430252640880428, "frac_reward_zero_std": 0.0, "grad_norm": 3.663283586502075, "learning_rate": 6.660606060606061e-06, "loss": -0.0174, "num_tokens": 2492366.0, "reward": 0.49698805809020996, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.49698805809020996, "reward_meter_std": 0.368730753660202, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3687307834625244, "reward_total_composite_mean": 0.49698805809020996, "reward_total_composite_std": 0.368730753660202, "reward_total_mean": 0.49698805809020996, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.49698805809020996, "rewards/meter/std": 0.368730753660202, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.49698805809020996, "rewards/total_composite/std": 0.368730753660202, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9964687824249268, "sampling/importance_sampling_ratio/min": 0.25461694598197937, "sampling/sampling_logp_difference/max": 1.8465352058410645, "sampling/sampling_logp_difference/mean": 0.02970782294869423, "step": 1103 }, { "clip_ratio/high_max": 0.028512453194707632, "clip_ratio/high_mean": 0.028512453194707632, "clip_ratio/low_mean": 0.01027012919075787, "clip_ratio/low_min": 0.01027012919075787, "clip_ratio/region_mean": 0.0387825823854655, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 51.125, "completions/mean_terminated_length": 51.125, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.18264612928032875, "epoch": 0.04434269189058923, "frac_reward_zero_std": 0.0, "grad_norm": 7.544509410858154, "learning_rate": 6.657575757575758e-06, "loss": -0.0068, "num_tokens": 2494079.0, "reward": 0.6021931171417236, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6021931171417236, "reward_meter_std": 0.43061062693595886, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4306105971336365, "reward_total_composite_mean": 0.6021931171417236, "reward_total_composite_std": 0.43061062693595886, "reward_total_mean": 0.6021931171417236, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6021931171417236, "rewards/meter/std": 0.43061062693595886, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6021931171417236, "rewards/total_composite/std": 0.43061062693595886, "sampling/importance_sampling_ratio/max": 1.6187946796417236, "sampling/importance_sampling_ratio/mean": 0.9914664030075073, "sampling/importance_sampling_ratio/min": 0.24896980822086334, "sampling/sampling_logp_difference/max": 1.3904236555099487, "sampling/sampling_logp_difference/mean": 0.05377590283751488, "step": 1104 }, { "clip_ratio/high_max": 0.035976887214928865, "clip_ratio/high_mean": 0.035976887214928865, "clip_ratio/low_mean": 0.00491898157633841, "clip_ratio/low_min": 0.00491898157633841, "clip_ratio/region_mean": 0.040895868791267276, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 51.75, "completions/mean_terminated_length": 51.75, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.2174948314204812, "epoch": 0.044382857372374185, "frac_reward_zero_std": 0.0, "grad_norm": 8.104043960571289, "learning_rate": 6.654545454545455e-06, "loss": 0.0134, "num_tokens": 2495813.0, "reward": 0.8142023086547852, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8142023086547852, "reward_meter_std": 0.21253840625286102, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.21253842115402222, "reward_total_composite_mean": 0.8142023086547852, "reward_total_composite_std": 0.21253840625286102, "reward_total_mean": 0.8142023086547852, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8142023086547852, "rewards/meter/std": 0.21253840625286102, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8142023086547852, "rewards/total_composite/std": 0.21253840625286102, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0077879428863525, "sampling/importance_sampling_ratio/min": 0.188982293009758, "sampling/sampling_logp_difference/max": 1.6661019325256348, "sampling/sampling_logp_difference/mean": 0.06759097427129745, "step": 1105 }, { "clip_ratio/high_max": 0.00831845449283719, "clip_ratio/high_mean": 0.00831845449283719, "clip_ratio/low_mean": 0.00146484375, "clip_ratio/low_min": 0.00146484375, "clip_ratio/region_mean": 0.00978329824283719, "completions/clipped_ratio": 0.0, "completions/max_length": 301.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 275.375, "completions/mean_terminated_length": 275.375, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "entropy": 0.05004570330493152, "epoch": 0.04442302285415914, "frac_reward_zero_std": 0.0, "grad_norm": 1.9001294374465942, "learning_rate": 6.651515151515152e-06, "loss": -0.0234, "num_tokens": 2499568.0, "reward": 0.46502256393432617, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.05050762742757797, "reward_meter_mean": 0.8482787013053894, "reward_meter_std": 0.33679211139678955, "reward_repeat_penalty_mean": 0.6220238208770752, "reward_repeat_penalty_std": 0.03127124905586243, "reward_std": 0.1884276121854782, "reward_total_composite_mean": 0.46502256393432617, "reward_total_composite_std": 0.1884276121854782, "reward_total_mean": 0.46502256393432617, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.05050762742757797, "rewards/meter/mean": 0.8482787013053894, "rewards/meter/std": 0.33679211139678955, "rewards/repeat_penalty/mean": 0.6220238208770752, "rewards/repeat_penalty/std": 0.03127124905586243, "rewards/total_composite/mean": 0.46502256393432617, "rewards/total_composite/std": 0.1884276121854782, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0009040832519531, "sampling/importance_sampling_ratio/min": 0.171641543507576, "sampling/sampling_logp_difference/max": 1.7623469829559326, "sampling/sampling_logp_difference/mean": 0.012011334300041199, "step": 1106 }, { "clip_ratio/high_max": 0.009699730202555656, "clip_ratio/high_mean": 0.009699730202555656, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/region_mean": 0.012989203911274672, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 38.25, "completions/mean_terminated_length": 38.25, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.07856028527021408, "epoch": 0.04446318833594409, "frac_reward_zero_std": 0.0, "grad_norm": 2.1617753505706787, "learning_rate": 6.6484848484848485e-06, "loss": 0.0015, "num_tokens": 2501018.0, "reward": 0.92488694190979, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.92488694190979, "reward_meter_std": 0.09670114517211914, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09670114517211914, "reward_total_composite_mean": 0.92488694190979, "reward_total_composite_std": 0.09670114517211914, "reward_total_mean": 0.92488694190979, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.92488694190979, "rewards/meter/std": 0.09670114517211914, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.92488694190979, "rewards/total_composite/std": 0.09670114517211914, "sampling/importance_sampling_ratio/max": 1.7915105819702148, "sampling/importance_sampling_ratio/mean": 1.0063523054122925, "sampling/importance_sampling_ratio/min": 0.5178860425949097, "sampling/sampling_logp_difference/max": 0.6579999923706055, "sampling/sampling_logp_difference/mean": 0.012348960153758526, "step": 1107 }, { "clip_ratio/high_max": 0.015104167046956718, "clip_ratio/high_mean": 0.015104167046956718, "clip_ratio/low_mean": 0.014691225020214915, "clip_ratio/low_min": 0.014691225020214915, "clip_ratio/region_mean": 0.029795392067171633, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 79.25, "completions/mean_terminated_length": 79.25, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.2001291485503316, "epoch": 0.04450335381772905, "frac_reward_zero_std": 0.0, "grad_norm": 6.412082672119141, "learning_rate": 6.645454545454546e-06, "loss": -0.0307, "num_tokens": 2503004.0, "reward": 0.39024245738983154, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4985108971595764, "reward_meter_std": 0.402507483959198, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.17728103697299957, "reward_std": 0.3671255111694336, "reward_total_composite_mean": 0.39024245738983154, "reward_total_composite_std": 0.367125540971756, "reward_total_mean": 0.39024245738983154, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4985108971595764, "rewards/meter/std": 0.402507483959198, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.17728103697299957, "rewards/total_composite/mean": 0.39024245738983154, "rewards/total_composite/std": 0.367125540971756, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0030810832977295, "sampling/importance_sampling_ratio/min": 0.06005259230732918, "sampling/sampling_logp_difference/max": 2.8125345706939697, "sampling/sampling_logp_difference/mean": 0.043393637984991074, "step": 1108 }, { "clip_ratio/high_max": 0.02806122461333871, "clip_ratio/high_mean": 0.02806122461333871, "clip_ratio/low_mean": 0.01064311619848013, "clip_ratio/low_min": 0.01064311619848013, "clip_ratio/region_mean": 0.03870434081181884, "completions/clipped_ratio": 0.0, "completions/max_length": 49.0, "completions/max_terminated_length": 49.0, "completions/mean_length": 47.375, "completions/mean_terminated_length": 47.375, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.19667030405253172, "epoch": 0.044543519299514, "frac_reward_zero_std": 0.0, "grad_norm": 8.44660472869873, "learning_rate": 6.642424242424242e-06, "loss": -0.025, "num_tokens": 2504551.0, "reward": 0.9353389143943787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9353389143943787, "reward_meter_std": 0.06214950606226921, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.062149498611688614, "reward_total_composite_mean": 0.9353389143943787, "reward_total_composite_std": 0.06214950606226921, "reward_total_mean": 0.9353389143943787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9353389143943787, "rewards/meter/std": 0.06214950606226921, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9353389143943787, "rewards/total_composite/std": 0.06214950606226921, "sampling/importance_sampling_ratio/max": 1.534562110900879, "sampling/importance_sampling_ratio/mean": 0.9965187907218933, "sampling/importance_sampling_ratio/min": 0.21853457391262054, "sampling/sampling_logp_difference/max": 1.5208110809326172, "sampling/sampling_logp_difference/mean": 0.043186187744140625, "step": 1109 }, { "clip_ratio/high_max": 0.024372577434405684, "clip_ratio/high_mean": 0.024372577434405684, "clip_ratio/low_mean": 0.009535853168927133, "clip_ratio/low_min": 0.009535853168927133, "clip_ratio/region_mean": 0.03390843060333282, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 150.0, "completions/mean_terminated_length": 150.0, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.17426198534667492, "epoch": 0.044583684781298955, "frac_reward_zero_std": 0.0, "grad_norm": 4.039949893951416, "learning_rate": 6.63939393939394e-06, "loss": -0.022, "num_tokens": 2507175.0, "reward": 0.5778908729553223, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9286544322967529, "reward_meter_std": 0.16951557993888855, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.1044817790389061, "reward_total_composite_mean": 0.5778908729553223, "reward_total_composite_std": 0.1044817790389061, "reward_total_mean": 0.5778908729553223, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9286544322967529, "rewards/meter/std": 0.16951557993888855, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.5778908729553223, "rewards/total_composite/std": 0.1044817790389061, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9992915987968445, "sampling/importance_sampling_ratio/min": 0.14697086811065674, "sampling/sampling_logp_difference/max": 1.9175208806991577, "sampling/sampling_logp_difference/mean": 0.038920193910598755, "step": 1110 }, { "clip_ratio/high_max": 0.0051369862630963326, "clip_ratio/high_mean": 0.0051369862630963326, "clip_ratio/low_mean": 0.0050675676902756095, "clip_ratio/low_min": 0.0050675676902756095, "clip_ratio/region_mean": 0.010204553953371942, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.03971482953056693, "epoch": 0.04462385026308391, "frac_reward_zero_std": 0.0, "grad_norm": 3.789764881134033, "learning_rate": 6.6363636363636375e-06, "loss": 0.0099, "num_tokens": 2508931.0, "reward": 0.9711061716079712, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9711061716079712, "reward_meter_std": 0.02328452095389366, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.023284511640667915, "reward_total_composite_mean": 0.9711061716079712, "reward_total_composite_std": 0.02328452095389366, "reward_total_mean": 0.9711061716079712, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9711061716079712, "rewards/meter/std": 0.02328452095389366, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9711061716079712, "rewards/total_composite/std": 0.02328452095389366, "sampling/importance_sampling_ratio/max": 1.7336740493774414, "sampling/importance_sampling_ratio/mean": 1.0022356510162354, "sampling/importance_sampling_ratio/min": 0.2580508887767792, "sampling/sampling_logp_difference/max": 1.3545985221862793, "sampling/sampling_logp_difference/mean": 0.011047722771763802, "step": 1111 }, { "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/region_mean": 0.006189613603055477, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 106.0, "completions/mean_terminated_length": 106.0, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.01866412186063826, "epoch": 0.04466401574486886, "frac_reward_zero_std": 0.0, "grad_norm": 0.8946425914764404, "learning_rate": 6.633333333333334e-06, "loss": -0.0457, "num_tokens": 2511051.0, "reward": 0.7782148122787476, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9727685451507568, "reward_meter_std": 0.053206540644168854, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04256521537899971, "reward_total_composite_mean": 0.7782148122787476, "reward_total_composite_std": 0.04256521537899971, "reward_total_mean": 0.7782148122787476, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9727685451507568, "rewards/meter/std": 0.053206540644168854, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7782148122787476, "rewards/total_composite/std": 0.04256521537899971, "sampling/importance_sampling_ratio/max": 1.4477814435958862, "sampling/importance_sampling_ratio/mean": 0.9997245073318481, "sampling/importance_sampling_ratio/min": 0.5710649490356445, "sampling/sampling_logp_difference/max": 0.56025230884552, "sampling/sampling_logp_difference/mean": 0.0040269712917506695, "step": 1112 }, { "clip_ratio/high_max": 0.010064412374049425, "clip_ratio/high_mean": 0.010064412374049425, "clip_ratio/low_mean": 0.010869565419852734, "clip_ratio/low_min": 0.010869565419852734, "clip_ratio/region_mean": 0.02093397779390216, "completions/clipped_ratio": 0.0, "completions/max_length": 27.0, "completions/max_terminated_length": 27.0, "completions/mean_length": 23.5, "completions/mean_terminated_length": 23.5, "completions/min_length": 23.0, "completions/min_terminated_length": 23.0, "entropy": 0.08848860999569297, "epoch": 0.044704181226653816, "frac_reward_zero_std": 0.0, "grad_norm": 6.619121074676514, "learning_rate": 6.630303030303031e-06, "loss": -0.0282, "num_tokens": 2512391.0, "reward": 0.867714524269104, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.867714524269104, "reward_meter_std": 0.052471715956926346, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.052471719682216644, "reward_total_composite_mean": 0.867714524269104, "reward_total_composite_std": 0.052471715956926346, "reward_total_mean": 0.867714524269104, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.867714524269104, "rewards/meter/std": 0.052471715956926346, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.867714524269104, "rewards/total_composite/std": 0.052471715956926346, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008794903755188, "sampling/importance_sampling_ratio/min": 0.4641389846801758, "sampling/sampling_logp_difference/max": 0.767571210861206, "sampling/sampling_logp_difference/mean": 0.024627521634101868, "step": 1113 }, { "clip_ratio/high_max": 0.017787929391488433, "clip_ratio/high_mean": 0.017787929391488433, "clip_ratio/low_mean": 0.014953542733564973, "clip_ratio/low_min": 0.014953542733564973, "clip_ratio/region_mean": 0.032741472125053406, "completions/clipped_ratio": 0.0, "completions/max_length": 49.0, "completions/max_terminated_length": 49.0, "completions/mean_length": 43.0, "completions/mean_terminated_length": 43.0, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.15364955179393291, "epoch": 0.04474434670843877, "frac_reward_zero_std": 0.0, "grad_norm": 11.225849151611328, "learning_rate": 6.627272727272728e-06, "loss": -0.0256, "num_tokens": 2513999.0, "reward": 0.8079327940940857, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8079327940940857, "reward_meter_std": 0.1427396684885025, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1427396684885025, "reward_total_composite_mean": 0.8079327940940857, "reward_total_composite_std": 0.1427396684885025, "reward_total_mean": 0.8079327940940857, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8079327940940857, "rewards/meter/std": 0.1427396684885025, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8079327940940857, "rewards/total_composite/std": 0.1427396684885025, "sampling/importance_sampling_ratio/max": 1.9035590887069702, "sampling/importance_sampling_ratio/mean": 1.0110169649124146, "sampling/importance_sampling_ratio/min": 0.3785325586795807, "sampling/sampling_logp_difference/max": 0.9714531898498535, "sampling/sampling_logp_difference/mean": 0.04281027242541313, "step": 1114 }, { "clip_ratio/high_max": 0.029880503891035914, "clip_ratio/high_mean": 0.029880503891035914, "clip_ratio/low_mean": 0.00828598509542644, "clip_ratio/low_min": 0.00828598509542644, "clip_ratio/region_mean": 0.038166488986462355, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 49.125, "completions/mean_terminated_length": 49.125, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.1894742762669921, "epoch": 0.044784512190223724, "frac_reward_zero_std": 0.0, "grad_norm": 7.548732757568359, "learning_rate": 6.624242424242425e-06, "loss": -0.0173, "num_tokens": 2515720.0, "reward": 0.7826473712921143, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7826473712921143, "reward_meter_std": 0.3253748416900635, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3253748118877411, "reward_total_composite_mean": 0.7826473712921143, "reward_total_composite_std": 0.3253748416900635, "reward_total_mean": 0.7826473712921143, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7826473712921143, "rewards/meter/std": 0.3253748416900635, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7826473712921143, "rewards/total_composite/std": 0.3253748416900635, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001320719718933, "sampling/importance_sampling_ratio/min": 0.14259269833564758, "sampling/sampling_logp_difference/max": 1.9477629661560059, "sampling/sampling_logp_difference/mean": 0.051869362592697144, "step": 1115 }, { "clip_ratio/high_max": 0.0034246575087308884, "clip_ratio/high_mean": 0.0034246575087308884, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/region_mean": 0.0051369862630963326, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 73.125, "completions/mean_terminated_length": 73.125, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.03595073730684817, "epoch": 0.04482467767200868, "frac_reward_zero_std": 0.0, "grad_norm": 8.76555347442627, "learning_rate": 6.621212121212121e-06, "loss": 0.0026, "num_tokens": 2517481.0, "reward": 0.9871143698692322, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9871143698692322, "reward_meter_std": 0.010886327363550663, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010886335745453835, "reward_total_composite_mean": 0.9871143698692322, "reward_total_composite_std": 0.010886327363550663, "reward_total_mean": 0.9871143698692322, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9871143698692322, "rewards/meter/std": 0.010886327363550663, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9871143698692322, "rewards/total_composite/std": 0.010886327363550663, "sampling/importance_sampling_ratio/max": 1.276580810546875, "sampling/importance_sampling_ratio/mean": 0.997620165348053, "sampling/importance_sampling_ratio/min": 0.2160053551197052, "sampling/sampling_logp_difference/max": 1.53245210647583, "sampling/sampling_logp_difference/mean": 0.011934218928217888, "step": 1116 }, { "clip_ratio/high_max": 0.002917782054282725, "clip_ratio/high_mean": 0.002917782054282725, "clip_ratio/low_mean": 0.001091206620912999, "clip_ratio/low_min": 0.001091206620912999, "clip_ratio/region_mean": 0.004008988675195724, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 341.0, "completions/mean_terminated_length": 341.0, "completions/min_length": 335.0, "completions/min_terminated_length": 335.0, "entropy": 0.03286444069817662, "epoch": 0.04486484315379363, "frac_reward_zero_std": 0.0, "grad_norm": 0.39119431376457214, "learning_rate": 6.618181818181819e-06, "loss": 0.0033, "num_tokens": 2521793.0, "reward": 0.42904794216156006, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9725086688995361, "reward_meter_std": 0.01511977519840002, "reward_repeat_penalty_mean": 0.5882353186607361, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006670480594038963, "reward_total_composite_mean": 0.42904794216156006, "reward_total_composite_std": 0.006670483388006687, "reward_total_mean": 0.42904794216156006, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9725086688995361, "rewards/meter/std": 0.01511977519840002, "rewards/repeat_penalty/mean": 0.5882353186607361, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.42904794216156006, "rewards/total_composite/std": 0.006670483388006687, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016568899154663, "sampling/importance_sampling_ratio/min": 0.07224062830209732, "sampling/sampling_logp_difference/max": 2.6277527809143066, "sampling/sampling_logp_difference/mean": 0.00816288124769926, "step": 1117 }, { "clip_ratio/high_max": 0.020123523310758173, "clip_ratio/high_mean": 0.020123523310758173, "clip_ratio/low_mean": 0.0043270515743643045, "clip_ratio/low_min": 0.0043270515743643045, "clip_ratio/region_mean": 0.024450574885122478, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 86.75, "completions/mean_terminated_length": 86.75, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.11351043824106455, "epoch": 0.044905008635578586, "frac_reward_zero_std": 0.0, "grad_norm": 4.05780553817749, "learning_rate": 6.615151515151516e-06, "loss": -0.0082, "num_tokens": 2523775.0, "reward": 0.6824886798858643, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7943093776702881, "reward_meter_std": 0.21005631983280182, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.2206040620803833, "reward_total_composite_mean": 0.6824886798858643, "reward_total_composite_std": 0.2206040620803833, "reward_total_mean": 0.6824886798858643, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7943093776702881, "rewards/meter/std": 0.21005631983280182, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6824886798858643, "rewards/total_composite/std": 0.2206040620803833, "sampling/importance_sampling_ratio/max": 1.9720497131347656, "sampling/importance_sampling_ratio/mean": 0.9958962202072144, "sampling/importance_sampling_ratio/min": 0.3371534049510956, "sampling/sampling_logp_difference/max": 1.0872173309326172, "sampling/sampling_logp_difference/mean": 0.028331898152828217, "step": 1118 }, { "clip_ratio/high_max": 0.006250000325962901, "clip_ratio/high_mean": 0.006250000325962901, "clip_ratio/low_mean": 0.012297285255044699, "clip_ratio/low_min": 0.012297285255044699, "clip_ratio/region_mean": 0.0185472855810076, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 60.5, "completions/mean_terminated_length": 60.5, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.061238054651767015, "epoch": 0.04494517411736354, "frac_reward_zero_std": 0.0, "grad_norm": 4.082438945770264, "learning_rate": 6.612121212121213e-06, "loss": 0.0076, "num_tokens": 2525659.0, "reward": 0.9949465990066528, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949465990066528, "reward_meter_std": 0.0008418544312007725, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008418540237471461, "reward_total_composite_mean": 0.9949465990066528, "reward_total_composite_std": 0.0008418544312007725, "reward_total_mean": 0.9949465990066528, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949465990066528, "rewards/meter/std": 0.0008418544312007725, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949465990066528, "rewards/total_composite/std": 0.0008418544312007725, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0043621063232422, "sampling/importance_sampling_ratio/min": 0.4875824749469757, "sampling/sampling_logp_difference/max": 1.0732321739196777, "sampling/sampling_logp_difference/mean": 0.016099898144602776, "step": 1119 }, { "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/region_mean": 0.004099462414160371, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 60.875, "completions/mean_terminated_length": 60.875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.04563273023813963, "epoch": 0.044985339599148494, "frac_reward_zero_std": 0.0, "grad_norm": 3.3688883781433105, "learning_rate": 6.609090909090909e-06, "loss": -0.0067, "num_tokens": 2527370.0, "reward": 0.9956821203231812, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956821203231812, "reward_meter_std": 0.0002610879309941083, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002611094678286463, "reward_total_composite_mean": 0.9956821203231812, "reward_total_composite_std": 0.0002610879309941083, "reward_total_mean": 0.9956821203231812, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956821203231812, "rewards/meter/std": 0.0002610879309941083, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956821203231812, "rewards/total_composite/std": 0.0002610879309941083, "sampling/importance_sampling_ratio/max": 1.5681954622268677, "sampling/importance_sampling_ratio/mean": 1.0035232305526733, "sampling/importance_sampling_ratio/min": 0.3273591101169586, "sampling/sampling_logp_difference/max": 1.1166975498199463, "sampling/sampling_logp_difference/mean": 0.012183277867734432, "step": 1120 }, { "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/low_mean": 0.005462184897623956, "clip_ratio/low_min": 0.005462184897623956, "clip_ratio/region_mean": 0.007198296021670103, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 69.75, "completions/mean_terminated_length": 69.75, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.061232382198795676, "epoch": 0.04502550508093345, "frac_reward_zero_std": 0.0, "grad_norm": 6.166923999786377, "learning_rate": 6.606060606060607e-06, "loss": -0.0073, "num_tokens": 2529160.0, "reward": 0.3661602735519409, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7864480018615723, "reward_meter_std": 0.26775890588760376, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.1511857956647873, "reward_std": 0.13910023868083954, "reward_total_composite_mean": 0.3661602735519409, "reward_total_composite_std": 0.13910025358200073, "reward_total_mean": 0.3661602735519409, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7864480018615723, "rewards/meter/std": 0.26775890588760376, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.1511857956647873, "rewards/total_composite/mean": 0.3661602735519409, "rewards/total_composite/std": 0.13910025358200073, "sampling/importance_sampling_ratio/max": 1.638248085975647, "sampling/importance_sampling_ratio/mean": 1.0017613172531128, "sampling/importance_sampling_ratio/min": 0.219312384724617, "sampling/sampling_logp_difference/max": 1.5172581672668457, "sampling/sampling_logp_difference/mean": 0.01315419189631939, "step": 1121 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 73.0, "completions/mean_terminated_length": 73.0, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.021201700437813997, "epoch": 0.0450656705627184, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.603030303030303e-06, "loss": 0.0, "num_tokens": 2530984.0, "reward": 0.9929870963096619, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9929870963096619, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9929870963096619, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9929870963096619, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9929870963096619, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929870963096619, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0893265008926392, "sampling/importance_sampling_ratio/mean": 1.000472068786621, "sampling/importance_sampling_ratio/min": 0.7952372431755066, "sampling/sampling_logp_difference/max": 0.229114830493927, "sampling/sampling_logp_difference/mean": 0.0027792660985141993, "step": 1122 }, { "clip_ratio/high_max": 0.0149312699213624, "clip_ratio/high_mean": 0.0149312699213624, "clip_ratio/low_mean": 0.01133623847272247, "clip_ratio/low_min": 0.01133623847272247, "clip_ratio/region_mean": 0.02626750839408487, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.198346434161067, "epoch": 0.045105836044503356, "frac_reward_zero_std": 0.0, "grad_norm": 4.8726959228515625, "learning_rate": 6.600000000000001e-06, "loss": -0.0032, "num_tokens": 2532779.0, "reward": 0.996692419052124, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996692419052124, "reward_meter_std": 0.0006456730188801885, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006456732517108321, "reward_total_composite_mean": 0.996692419052124, "reward_total_composite_std": 0.0006456730188801885, "reward_total_mean": 0.996692419052124, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996692419052124, "rewards/meter/std": 0.0006456730188801885, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996692419052124, "rewards/total_composite/std": 0.0006456730188801885, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9999663233757019, "sampling/importance_sampling_ratio/min": 0.16886253654956818, "sampling/sampling_logp_difference/max": 1.778670310974121, "sampling/sampling_logp_difference/mean": 0.04234900325536728, "step": 1123 }, { "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/low_mean": 0.006849315017461777, "clip_ratio/low_min": 0.006849315017461777, "clip_ratio/region_mean": 0.01032153726555407, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.875, "completions/mean_terminated_length": 72.875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.028838476166129112, "epoch": 0.04514600152628831, "frac_reward_zero_std": 0.0, "grad_norm": 0.6933730244636536, "learning_rate": 6.596969696969698e-06, "loss": 0.0006, "num_tokens": 2534594.0, "reward": 0.9930577874183655, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9930577874183655, "reward_meter_std": 0.0013781489105895162, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001378138200379908, "reward_total_composite_mean": 0.9930577874183655, "reward_total_composite_std": 0.0013781489105895162, "reward_total_mean": 0.9930577874183655, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9930577874183655, "rewards/meter/std": 0.0013781489105895162, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9930577874183655, "rewards/total_composite/std": 0.0013781489105895162, "sampling/importance_sampling_ratio/max": 1.413206696510315, "sampling/importance_sampling_ratio/mean": 1.000672459602356, "sampling/importance_sampling_ratio/min": 0.6142716407775879, "sampling/sampling_logp_difference/max": 0.48731809854507446, "sampling/sampling_logp_difference/mean": 0.004648053552955389, "step": 1124 }, { "clip_ratio/high_max": 0.010517970658838749, "clip_ratio/high_mean": 0.010517970658838749, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.010517970658838749, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.375, "completions/mean_terminated_length": 70.375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.051455921959131956, "epoch": 0.045186167008073264, "frac_reward_zero_std": 0.0, "grad_norm": 3.602660894393921, "learning_rate": 6.593939393939395e-06, "loss": -0.0499, "num_tokens": 2536357.0, "reward": 0.9664039611816406, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9664039611816406, "reward_meter_std": 0.08127888292074203, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.08127887547016144, "reward_total_composite_mean": 0.9664039611816406, "reward_total_composite_std": 0.08127888292074203, "reward_total_mean": 0.9664039611816406, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9664039611816406, "rewards/meter/std": 0.08127888292074203, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9664039611816406, "rewards/total_composite/std": 0.08127888292074203, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9998224377632141, "sampling/importance_sampling_ratio/min": 0.22992707788944244, "sampling/sampling_logp_difference/max": 1.4699931144714355, "sampling/sampling_logp_difference/mean": 0.015036996454000473, "step": 1125 }, { "clip_ratio/high_max": 0.02241882192902267, "clip_ratio/high_mean": 0.02241882192902267, "clip_ratio/low_mean": 0.007549191126599908, "clip_ratio/low_min": 0.007549191126599908, "clip_ratio/region_mean": 0.029968013055622578, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.15226891916245222, "epoch": 0.04522633248985822, "frac_reward_zero_std": 0.0, "grad_norm": 4.485559940338135, "learning_rate": 6.590909090909091e-06, "loss": -0.0067, "num_tokens": 2538137.0, "reward": 0.9959009289741516, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959009289741516, "reward_meter_std": 0.0018093610415235162, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0018093656981363893, "reward_total_composite_mean": 0.9959009289741516, "reward_total_composite_std": 0.0018093610415235162, "reward_total_mean": 0.9959009289741516, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959009289741516, "rewards/meter/std": 0.0018093610415235162, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9959009289741516, "rewards/total_composite/std": 0.0018093610415235162, "sampling/importance_sampling_ratio/max": 1.4944887161254883, "sampling/importance_sampling_ratio/mean": 0.997355580329895, "sampling/importance_sampling_ratio/min": 0.33068788051605225, "sampling/sampling_logp_difference/max": 1.1065802574157715, "sampling/sampling_logp_difference/mean": 0.030639950186014175, "step": 1126 }, { "clip_ratio/high_max": 0.01383362547494471, "clip_ratio/high_mean": 0.01383362547494471, "clip_ratio/low_mean": 0.006355932215228677, "clip_ratio/low_min": 0.006355932215228677, "clip_ratio/region_mean": 0.020189557690173388, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 60.625, "completions/mean_terminated_length": 60.625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.08485704753547907, "epoch": 0.04526649797164317, "frac_reward_zero_std": 0.0, "grad_norm": 3.99946928024292, "learning_rate": 6.5878787878787885e-06, "loss": -0.0117, "num_tokens": 2539806.0, "reward": 0.9501904249191284, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9501904249191284, "reward_meter_std": 0.06670597940683365, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06670597940683365, "reward_total_composite_mean": 0.9501904249191284, "reward_total_composite_std": 0.06670597940683365, "reward_total_mean": 0.9501904249191284, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9501904249191284, "rewards/meter/std": 0.06670597940683365, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9501904249191284, "rewards/total_composite/std": 0.06670597940683365, "sampling/importance_sampling_ratio/max": 1.6531167030334473, "sampling/importance_sampling_ratio/mean": 0.9991652369499207, "sampling/importance_sampling_ratio/min": 0.24098443984985352, "sampling/sampling_logp_difference/max": 1.423022985458374, "sampling/sampling_logp_difference/mean": 0.022387670353055, "step": 1127 }, { "clip_ratio/high_max": 0.006048386916518211, "clip_ratio/high_mean": 0.006048386916518211, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/region_mean": 0.008131720358505845, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.375, "completions/mean_terminated_length": 61.375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.05297265062108636, "epoch": 0.045306663453428125, "frac_reward_zero_std": 0.0, "grad_norm": 1.894775390625, "learning_rate": 6.584848484848485e-06, "loss": -0.0059, "num_tokens": 2541433.0, "reward": 0.9959591627120972, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959591627120972, "reward_meter_std": 0.0002977726107928902, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002977752883452922, "reward_total_composite_mean": 0.9959591627120972, "reward_total_composite_std": 0.0002977726107928902, "reward_total_mean": 0.9959591627120972, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959591627120972, "rewards/meter/std": 0.0002977726107928902, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9959591627120972, "rewards/total_composite/std": 0.0002977726107928902, "sampling/importance_sampling_ratio/max": 1.208601951599121, "sampling/importance_sampling_ratio/mean": 1.0015958547592163, "sampling/importance_sampling_ratio/min": 0.5194104909896851, "sampling/sampling_logp_difference/max": 0.6550607681274414, "sampling/sampling_logp_difference/mean": 0.008953627198934555, "step": 1128 }, { "clip_ratio/high_max": 0.0071450776886194944, "clip_ratio/high_mean": 0.0071450776886194944, "clip_ratio/low_mean": 0.005044843070209026, "clip_ratio/low_min": 0.005044843070209026, "clip_ratio/region_mean": 0.01218992075882852, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 375.25, "completions/mean_terminated_length": 238.5, "completions/min_length": 221.0, "completions/min_terminated_length": 221.0, "entropy": 0.09304305911064148, "epoch": 0.04534682893521308, "frac_reward_zero_std": 0.0, "grad_norm": 1.491158366203308, "learning_rate": 6.581818181818182e-06, "loss": -0.2408, "num_tokens": 2544019.0, "reward": 0.16618868708610535, "reward_arabic_clean_mean": 0.5, "reward_arabic_clean_std": 0.5345224738121033, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.2777460217475891, "reward_meter_mean": 0.367766410112381, "reward_meter_std": 0.4184606671333313, "reward_repeat_penalty_mean": 0.7755848169326782, "reward_repeat_penalty_std": 0.2187419831752777, "reward_std": 0.2396627515554428, "reward_total_composite_mean": 0.16618868708610535, "reward_total_composite_std": 0.2396627515554428, "reward_total_mean": 0.16618868708610535, "rewards/arabic_clean/mean": 0.5, "rewards/arabic_clean/std": 0.5345224738121033, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.2777460217475891, "rewards/meter/mean": 0.367766410112381, "rewards/meter/std": 0.4184606671333313, "rewards/repeat_penalty/mean": 0.7755848169326782, "rewards/repeat_penalty/std": 0.2187419831752777, "rewards/total_composite/mean": 0.16618868708610535, "rewards/total_composite/std": 0.2396627515554428, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0069185495376587, "sampling/importance_sampling_ratio/min": 0.004834904335439205, "sampling/sampling_logp_difference/max": 5.3318939208984375, "sampling/sampling_logp_difference/mean": 0.03679489716887474, "step": 1129 }, { "clip_ratio/high_max": 0.030356566421687603, "clip_ratio/high_mean": 0.030356566421687603, "clip_ratio/low_mean": 0.015434782486408949, "clip_ratio/low_min": 0.015434782486408949, "clip_ratio/region_mean": 0.04579134890809655, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 108.5, "completions/mean_terminated_length": 50.85714340209961, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.25371737964451313, "epoch": 0.04538699441699803, "frac_reward_zero_std": 0.0, "grad_norm": 3.4096953868865967, "learning_rate": 6.578787878787879e-06, "loss": -0.0873, "num_tokens": 2545823.0, "reward": 0.47555798292160034, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5511125326156616, "reward_meter_std": 0.42126619815826416, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.38399630784988403, "reward_total_composite_mean": 0.47555798292160034, "reward_total_composite_std": 0.38399630784988403, "reward_total_mean": 0.47555798292160034, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5511125326156616, "rewards/meter/std": 0.42126619815826416, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.47555798292160034, "rewards/total_composite/std": 0.38399630784988403, "sampling/importance_sampling_ratio/max": 1.7676554918289185, "sampling/importance_sampling_ratio/mean": 1.007746696472168, "sampling/importance_sampling_ratio/min": 0.24680811166763306, "sampling/sampling_logp_difference/max": 1.399144172668457, "sampling/sampling_logp_difference/mean": 0.054451193660497665, "step": 1130 }, { "clip_ratio/high_max": 0.013678450835868716, "clip_ratio/high_mean": 0.013678450835868716, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/region_mean": 0.02617845102213323, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 54.125, "completions/mean_terminated_length": 54.125, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.08877178048714995, "epoch": 0.04542715989878299, "frac_reward_zero_std": 0.0, "grad_norm": 8.209676742553711, "learning_rate": 6.575757575757577e-06, "loss": -0.0297, "num_tokens": 2547552.0, "reward": 0.9417009353637695, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9417009353637695, "reward_meter_std": 0.008068003691732883, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008068017661571503, "reward_total_composite_mean": 0.9417009353637695, "reward_total_composite_std": 0.008068003691732883, "reward_total_mean": 0.9417009353637695, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9417009353637695, "rewards/meter/std": 0.008068003691732883, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9417009353637695, "rewards/total_composite/std": 0.008068003691732883, "sampling/importance_sampling_ratio/max": 1.376886010169983, "sampling/importance_sampling_ratio/mean": 0.9954664707183838, "sampling/importance_sampling_ratio/min": 0.08513900637626648, "sampling/sampling_logp_difference/max": 2.463469982147217, "sampling/sampling_logp_difference/mean": 0.02913379855453968, "step": 1131 }, { "clip_ratio/high_max": 0.05365684116259217, "clip_ratio/high_mean": 0.05365684116259217, "clip_ratio/low_mean": 0.014423077460378408, "clip_ratio/low_min": 0.014423077460378408, "clip_ratio/region_mean": 0.06807991862297058, "completions/clipped_ratio": 0.0, "completions/max_length": 31.0, "completions/max_terminated_length": 31.0, "completions/mean_length": 29.25, "completions/mean_terminated_length": 29.25, "completions/min_length": 26.0, "completions/min_terminated_length": 26.0, "entropy": 0.42541524581611156, "epoch": 0.04546732538056794, "frac_reward_zero_std": 0.0, "grad_norm": 14.512413024902344, "learning_rate": 6.572727272727273e-06, "loss": -0.0471, "num_tokens": 2548978.0, "reward": 0.8743196725845337, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8743196725845337, "reward_meter_std": 0.20709700882434845, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.20709700882434845, "reward_total_composite_mean": 0.8743196725845337, "reward_total_composite_std": 0.20709700882434845, "reward_total_mean": 0.8743196725845337, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8743196725845337, "rewards/meter/std": 0.20709700882434845, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8743196725845337, "rewards/total_composite/std": 0.20709700882434845, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0037531852722168, "sampling/importance_sampling_ratio/min": 0.0015519780572503805, "sampling/sampling_logp_difference/max": 6.468225002288818, "sampling/sampling_logp_difference/mean": 0.09784162044525146, "step": 1132 }, { "clip_ratio/high_max": 0.004167824285104871, "clip_ratio/high_mean": 0.004167824285104871, "clip_ratio/low_mean": 0.006215847097337246, "clip_ratio/low_min": 0.006215847097337246, "clip_ratio/region_mean": 0.010383671382442117, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 59.75, "completions/mean_terminated_length": 59.75, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.046739939134567976, "epoch": 0.045507490862352895, "frac_reward_zero_std": 0.0, "grad_norm": 1.9009183645248413, "learning_rate": 6.56969696969697e-06, "loss": 0.0098, "num_tokens": 2550664.0, "reward": 0.9870603084564209, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9870603084564209, "reward_meter_std": 0.004516107961535454, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004516112618148327, "reward_total_composite_mean": 0.9870603084564209, "reward_total_composite_std": 0.004516107961535454, "reward_total_mean": 0.9870603084564209, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9870603084564209, "rewards/meter/std": 0.004516107961535454, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9870603084564209, "rewards/total_composite/std": 0.004516107961535454, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0032687187194824, "sampling/importance_sampling_ratio/min": 0.4231666624546051, "sampling/sampling_logp_difference/max": 1.056100606918335, "sampling/sampling_logp_difference/mean": 0.013605736196041107, "step": 1133 }, { "clip_ratio/high_max": 0.026770169381052256, "clip_ratio/high_mean": 0.026770169381052256, "clip_ratio/low_mean": 0.010638833045959473, "clip_ratio/low_min": 0.010638833045959473, "clip_ratio/region_mean": 0.03740900242701173, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 70.25, "completions/mean_terminated_length": 70.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.18766964972019196, "epoch": 0.04554765634413785, "frac_reward_zero_std": 0.0, "grad_norm": 4.714395523071289, "learning_rate": 6.566666666666667e-06, "loss": 0.0062, "num_tokens": 2552346.0, "reward": 0.9971396923065186, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971396923065186, "reward_meter_std": 0.0017616376280784607, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0017616377444937825, "reward_total_composite_mean": 0.9971396923065186, "reward_total_composite_std": 0.0017616376280784607, "reward_total_mean": 0.9971396923065186, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971396923065186, "rewards/meter/std": 0.0017616376280784607, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971396923065186, "rewards/total_composite/std": 0.0017616376280784607, "sampling/importance_sampling_ratio/max": 1.841591238975525, "sampling/importance_sampling_ratio/mean": 1.0078562498092651, "sampling/importance_sampling_ratio/min": 0.2614503800868988, "sampling/sampling_logp_difference/max": 1.3415107727050781, "sampling/sampling_logp_difference/mean": 0.030578911304473877, "step": 1134 }, { "clip_ratio/high_max": 0.011680196272209287, "clip_ratio/high_mean": 0.011680196272209287, "clip_ratio/low_mean": 0.00811688310932368, "clip_ratio/low_min": 0.00811688310932368, "clip_ratio/region_mean": 0.019797079381532967, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 76.5, "completions/mean_terminated_length": 76.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.10036304593086243, "epoch": 0.0455878218259228, "frac_reward_zero_std": 0.0, "grad_norm": 3.5454463958740234, "learning_rate": 6.563636363636364e-06, "loss": 0.0005, "num_tokens": 2554294.0, "reward": 0.9800901412963867, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9800901412963867, "reward_meter_std": 0.01382446475327015, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.013824466615915298, "reward_total_composite_mean": 0.9800901412963867, "reward_total_composite_std": 0.01382446475327015, "reward_total_mean": 0.9800901412963867, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9800901412963867, "rewards/meter/std": 0.01382446475327015, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9800901412963867, "rewards/total_composite/std": 0.01382446475327015, "sampling/importance_sampling_ratio/max": 1.7477397918701172, "sampling/importance_sampling_ratio/mean": 1.005315899848938, "sampling/importance_sampling_ratio/min": 0.44200506806373596, "sampling/sampling_logp_difference/max": 0.8164339065551758, "sampling/sampling_logp_difference/mean": 0.01574569195508957, "step": 1135 }, { "clip_ratio/high_max": 0.005753153818659484, "clip_ratio/high_mean": 0.005753153818659484, "clip_ratio/low_mean": 0.010929276293609291, "clip_ratio/low_min": 0.010929276293609291, "clip_ratio/region_mean": 0.016682430112268776, "completions/clipped_ratio": 0.0, "completions/max_length": 427.0, "completions/max_terminated_length": 427.0, "completions/mean_length": 388.375, "completions/mean_terminated_length": 388.375, "completions/min_length": 362.0, "completions/min_terminated_length": 362.0, "entropy": 0.21599608566612005, "epoch": 0.04562798730770776, "frac_reward_zero_std": 0.0, "grad_norm": 2.327674627304077, "learning_rate": 6.56060606060606e-06, "loss": -0.019, "num_tokens": 2559265.0, "reward": 0.49994567036628723, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7410714626312256, "reward_count_adherence_std": 0.03696778044104576, "reward_meter_mean": 0.952599823474884, "reward_meter_std": 0.09515554457902908, "reward_repeat_penalty_mean": 0.716478705406189, "reward_repeat_penalty_std": 0.12055735290050507, "reward_std": 0.060885630548000336, "reward_total_composite_mean": 0.49994567036628723, "reward_total_composite_std": 0.06088562682271004, "reward_total_mean": 0.49994567036628723, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7410714626312256, "rewards/count_adherence/std": 0.03696778044104576, "rewards/meter/mean": 0.952599823474884, "rewards/meter/std": 0.09515554457902908, "rewards/repeat_penalty/mean": 0.716478705406189, "rewards/repeat_penalty/std": 0.12055735290050507, "rewards/total_composite/mean": 0.49994567036628723, "rewards/total_composite/std": 0.06088562682271004, "sampling/importance_sampling_ratio/max": 1.8240277767181396, "sampling/importance_sampling_ratio/mean": 1.0076520442962646, "sampling/importance_sampling_ratio/min": 0.18411126732826233, "sampling/sampling_logp_difference/max": 1.6922149658203125, "sampling/sampling_logp_difference/mean": 0.02755453996360302, "step": 1136 }, { "clip_ratio/high_max": 0.010732458787970245, "clip_ratio/high_mean": 0.010732458787970245, "clip_ratio/low_mean": 0.0009057971183210611, "clip_ratio/low_min": 0.0009057971183210611, "clip_ratio/region_mean": 0.011638255906291306, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 111.125, "completions/mean_terminated_length": 111.125, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.08285832032561302, "epoch": 0.04566815278949271, "frac_reward_zero_std": 0.0, "grad_norm": 3.7329790592193604, "learning_rate": 6.5575757575757585e-06, "loss": 0.0836, "num_tokens": 2561586.0, "reward": 0.7466674447059631, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.9837856292724609, "reward_meter_std": 0.012296498753130436, "reward_repeat_penalty_mean": 0.7892857789993286, "reward_repeat_penalty_std": 0.030304575338959694, "reward_std": 0.11019132286310196, "reward_total_composite_mean": 0.7466674447059631, "reward_total_composite_std": 0.11019133776426315, "reward_total_mean": 0.7466674447059631, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.9837856292724609, "rewards/meter/std": 0.012296498753130436, "rewards/repeat_penalty/mean": 0.7892857789993286, "rewards/repeat_penalty/std": 0.030304575338959694, "rewards/total_composite/mean": 0.7466674447059631, "rewards/total_composite/std": 0.11019133776426315, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9983324408531189, "sampling/importance_sampling_ratio/min": 0.13253353536128998, "sampling/sampling_logp_difference/max": 2.0209195613861084, "sampling/sampling_logp_difference/mean": 0.020305845886468887, "step": 1137 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 59.25, "completions/mean_terminated_length": 59.25, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.020128208212554455, "epoch": 0.045708318271277665, "frac_reward_zero_std": 0.0, "grad_norm": 2.7455623149871826, "learning_rate": 6.554545454545455e-06, "loss": -0.0078, "num_tokens": 2563284.0, "reward": 0.9892421960830688, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9892421960830688, "reward_meter_std": 0.0002119775745086372, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000211986611247994, "reward_total_composite_mean": 0.9892421960830688, "reward_total_composite_std": 0.0002119775745086372, "reward_total_mean": 0.9892421960830688, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9892421960830688, "rewards/meter/std": 0.0002119775745086372, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9892421960830688, "rewards/total_composite/std": 0.0002119775745086372, "sampling/importance_sampling_ratio/max": 1.097350001335144, "sampling/importance_sampling_ratio/mean": 1.0011262893676758, "sampling/importance_sampling_ratio/min": 0.960770845413208, "sampling/sampling_logp_difference/max": 0.09289813041687012, "sampling/sampling_logp_difference/mean": 0.002105327555909753, "step": 1138 }, { "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/low_mean": 0.008266184711828828, "clip_ratio/low_min": 0.008266184711828828, "clip_ratio/region_mean": 0.010282313684001565, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 60.875, "completions/mean_terminated_length": 60.875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.04918731888756156, "epoch": 0.04574848375306262, "frac_reward_zero_std": 0.0, "grad_norm": 3.7970211505889893, "learning_rate": 6.551515151515152e-06, "loss": -0.0093, "num_tokens": 2564995.0, "reward": 0.989832878112793, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.989832878112793, "reward_meter_std": 0.0006110123940743506, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006110109388828278, "reward_total_composite_mean": 0.989832878112793, "reward_total_composite_std": 0.0006110123940743506, "reward_total_mean": 0.989832878112793, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.989832878112793, "rewards/meter/std": 0.0006110123940743506, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.989832878112793, "rewards/total_composite/std": 0.0006110123940743506, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028979778289795, "sampling/importance_sampling_ratio/min": 0.28789207339286804, "sampling/sampling_logp_difference/max": 1.2451696395874023, "sampling/sampling_logp_difference/mean": 0.011983020231127739, "step": 1139 }, { "clip_ratio/high_max": 0.00657894741743803, "clip_ratio/high_mean": 0.00657894741743803, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.00889376224949956, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 56.25, "completions/mean_terminated_length": 56.25, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.061810399405658245, "epoch": 0.04578864923484757, "frac_reward_zero_std": 0.0, "grad_norm": 5.715131759643555, "learning_rate": 6.5484848484848494e-06, "loss": -0.0165, "num_tokens": 2566645.0, "reward": 0.2350359857082367, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2350359857082367, "reward_meter_std": 0.07872093468904495, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.07872093468904495, "reward_total_composite_mean": 0.2350359857082367, "reward_total_composite_std": 0.07872093468904495, "reward_total_mean": 0.2350359857082367, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2350359857082367, "rewards/meter/std": 0.07872093468904495, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2350359857082367, "rewards/total_composite/std": 0.07872093468904495, "sampling/importance_sampling_ratio/max": 1.7906800508499146, "sampling/importance_sampling_ratio/mean": 1.001919150352478, "sampling/importance_sampling_ratio/min": 0.6520004272460938, "sampling/sampling_logp_difference/max": 0.5825954675674438, "sampling/sampling_logp_difference/mean": 0.012172389775514603, "step": 1140 }, { "clip_ratio/high_max": 0.02358490601181984, "clip_ratio/high_mean": 0.02358490601181984, "clip_ratio/low_mean": 0.016250555869191885, "clip_ratio/low_min": 0.016250555869191885, "clip_ratio/region_mean": 0.039835461881011724, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 53.625, "completions/mean_terminated_length": 53.625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.1192027423530817, "epoch": 0.04582881471663253, "frac_reward_zero_std": 0.0, "grad_norm": 7.939047813415527, "learning_rate": 6.545454545454546e-06, "loss": 0.0143, "num_tokens": 2568290.0, "reward": 0.9498906135559082, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9498906135559082, "reward_meter_std": 0.007521784398704767, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007521784398704767, "reward_total_composite_mean": 0.9498906135559082, "reward_total_composite_std": 0.007521784398704767, "reward_total_mean": 0.9498906135559082, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9498906135559082, "rewards/meter/std": 0.007521784398704767, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9498906135559082, "rewards/total_composite/std": 0.007521784398704767, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9988313913345337, "sampling/importance_sampling_ratio/min": 0.08954470604658127, "sampling/sampling_logp_difference/max": 2.4130172729492188, "sampling/sampling_logp_difference/mean": 0.04453643411397934, "step": 1141 }, { "clip_ratio/high_max": 0.025911615695804358, "clip_ratio/high_mean": 0.025911615695804358, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/region_mean": 0.029036615742370486, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 39.125, "completions/mean_terminated_length": 39.125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.17795206978917122, "epoch": 0.04586898019841748, "frac_reward_zero_std": 0.0, "grad_norm": 4.282461166381836, "learning_rate": 6.542424242424243e-06, "loss": 0.0038, "num_tokens": 2569755.0, "reward": 0.9843659400939941, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9843659400939941, "reward_meter_std": 0.01807660609483719, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.018076593056321144, "reward_total_composite_mean": 0.9843659400939941, "reward_total_composite_std": 0.01807660609483719, "reward_total_mean": 0.9843659400939941, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9843659400939941, "rewards/meter/std": 0.01807660609483719, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9843659400939941, "rewards/total_composite/std": 0.01807660609483719, "sampling/importance_sampling_ratio/max": 1.3938076496124268, "sampling/importance_sampling_ratio/mean": 0.9939477443695068, "sampling/importance_sampling_ratio/min": 0.29377633333206177, "sampling/sampling_logp_difference/max": 1.2249364852905273, "sampling/sampling_logp_difference/mean": 0.03325369209051132, "step": 1142 }, { "clip_ratio/high_max": 0.011979256058111787, "clip_ratio/high_mean": 0.011979256058111787, "clip_ratio/low_mean": 0.004563539958326146, "clip_ratio/low_min": 0.004563539958326146, "clip_ratio/region_mean": 0.016542796016437933, "completions/clipped_ratio": 0.0, "completions/max_length": 498.0, "completions/max_terminated_length": 498.0, "completions/mean_length": 441.125, "completions/mean_terminated_length": 441.125, "completions/min_length": 400.0, "completions/min_terminated_length": 400.0, "entropy": 0.1385105513036251, "epoch": 0.045909145680202434, "frac_reward_zero_std": 0.0, "grad_norm": 1.155205488204956, "learning_rate": 6.5393939393939395e-06, "loss": -0.0509, "num_tokens": 2575500.0, "reward": 0.35644832253456116, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6323529481887817, "reward_count_adherence_std": 0.027230001986026764, "reward_meter_mean": 0.9610804319381714, "reward_meter_std": 0.050803039222955704, "reward_repeat_penalty_mean": 0.5833333134651184, "reward_repeat_penalty_std": 0.19633837044239044, "reward_std": 0.13268794119358063, "reward_total_composite_mean": 0.35644832253456116, "reward_total_composite_std": 0.13268794119358063, "reward_total_mean": 0.35644832253456116, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6323529481887817, "rewards/count_adherence/std": 0.027230001986026764, "rewards/meter/mean": 0.9610804319381714, "rewards/meter/std": 0.050803039222955704, "rewards/repeat_penalty/mean": 0.5833333134651184, "rewards/repeat_penalty/std": 0.19633837044239044, "rewards/total_composite/mean": 0.35644832253456116, "rewards/total_composite/std": 0.13268794119358063, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004508376121521, "sampling/importance_sampling_ratio/min": 0.28720539808273315, "sampling/sampling_logp_difference/max": 1.2659144401550293, "sampling/sampling_logp_difference/mean": 0.021479658782482147, "step": 1143 }, { "clip_ratio/high_max": 0.005727101757656783, "clip_ratio/high_mean": 0.005727101757656783, "clip_ratio/low_mean": 0.0020058397494722158, "clip_ratio/low_min": 0.0020058397494722158, "clip_ratio/region_mean": 0.007732941507128999, "completions/clipped_ratio": 0.0, "completions/max_length": 420.0, "completions/max_terminated_length": 420.0, "completions/mean_length": 387.75, "completions/mean_terminated_length": 387.75, "completions/min_length": 354.0, "completions/min_terminated_length": 354.0, "entropy": 0.06274456530809402, "epoch": 0.04594931116198739, "frac_reward_zero_std": 0.0, "grad_norm": 0.9518696665763855, "learning_rate": 6.536363636363638e-06, "loss": -0.0143, "num_tokens": 2580194.0, "reward": 0.36780261993408203, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7053571939468384, "reward_count_adherence_std": 0.0707879438996315, "reward_meter_mean": 0.9527665376663208, "reward_meter_std": 0.0463380366563797, "reward_repeat_penalty_mean": 0.5333124399185181, "reward_repeat_penalty_std": 0.15502820909023285, "reward_std": 0.1357763111591339, "reward_total_composite_mean": 0.36780261993408203, "reward_total_composite_std": 0.1357763111591339, "reward_total_mean": 0.36780261993408203, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7053571939468384, "rewards/count_adherence/std": 0.0707879438996315, "rewards/meter/mean": 0.9527665376663208, "rewards/meter/std": 0.0463380366563797, "rewards/repeat_penalty/mean": 0.5333124399185181, "rewards/repeat_penalty/std": 0.15502820909023285, "rewards/total_composite/mean": 0.36780261993408203, "rewards/total_composite/std": 0.1357763111591339, "sampling/importance_sampling_ratio/max": 1.8800609111785889, "sampling/importance_sampling_ratio/mean": 1.0018688440322876, "sampling/importance_sampling_ratio/min": 0.02310972288250923, "sampling/sampling_logp_difference/max": 3.7675018310546875, "sampling/sampling_logp_difference/mean": 0.011555654928088188, "step": 1144 }, { "clip_ratio/high_max": 0.0031250000465661287, "clip_ratio/high_mean": 0.0031250000465661287, "clip_ratio/low_mean": 0.006580086657777429, "clip_ratio/low_min": 0.006580086657777429, "clip_ratio/region_mean": 0.009705086704343557, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 77.625, "completions/mean_terminated_length": 77.625, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.13493561558425426, "epoch": 0.04598947664377234, "frac_reward_zero_std": 0.0, "grad_norm": 3.6901073455810547, "learning_rate": 6.533333333333334e-06, "loss": -0.0246, "num_tokens": 2582143.0, "reward": 0.9917942881584167, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9917942881584167, "reward_meter_std": 0.0048305438831448555, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004830518271774054, "reward_total_composite_mean": 0.9917942881584167, "reward_total_composite_std": 0.0048305438831448555, "reward_total_mean": 0.9917942881584167, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9917942881584167, "rewards/meter/std": 0.0048305438831448555, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917942881584167, "rewards/total_composite/std": 0.0048305438831448555, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0097332000732422, "sampling/importance_sampling_ratio/min": 0.34659311175346375, "sampling/sampling_logp_difference/max": 1.3061323165893555, "sampling/sampling_logp_difference/mean": 0.017006773501634598, "step": 1145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.0, "completions/mean_terminated_length": 72.0, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.016070799436420202, "epoch": 0.046029642125557296, "frac_reward_zero_std": 0.0, "grad_norm": 1.9219002723693848, "learning_rate": 6.530303030303031e-06, "loss": -0.0049, "num_tokens": 2583831.0, "reward": 0.9548450708389282, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963383078575134, "reward_meter_std": 0.0003733669000212103, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11756306141614914, "reward_total_composite_mean": 0.9548450708389282, "reward_total_composite_std": 0.11756306886672974, "reward_total_mean": 0.9548450708389282, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963383078575134, "rewards/meter/std": 0.0003733669000212103, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9548450708389282, "rewards/total_composite/std": 0.11756306886672974, "sampling/importance_sampling_ratio/max": 1.0752372741699219, "sampling/importance_sampling_ratio/mean": 1.000481128692627, "sampling/importance_sampling_ratio/min": 0.5831482410430908, "sampling/sampling_logp_difference/max": 0.539313793182373, "sampling/sampling_logp_difference/mean": 0.003385355230420828, "step": 1146 }, { "clip_ratio/high_max": 0.014285714365541935, "clip_ratio/high_mean": 0.014285714365541935, "clip_ratio/low_mean": 0.01907169120386243, "clip_ratio/low_min": 0.01907169120386243, "clip_ratio/region_mean": 0.033357405569404364, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 32.625, "completions/mean_terminated_length": 32.625, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.21602893620729446, "epoch": 0.04606980760734225, "frac_reward_zero_std": 0.0, "grad_norm": 15.72409725189209, "learning_rate": 6.527272727272728e-06, "loss": 0.0294, "num_tokens": 2585484.0, "reward": 0.89949631690979, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.89949631690979, "reward_meter_std": 0.18807001411914825, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.18807001411914825, "reward_total_composite_mean": 0.89949631690979, "reward_total_composite_std": 0.18807001411914825, "reward_total_mean": 0.89949631690979, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.89949631690979, "rewards/meter/std": 0.18807001411914825, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.89949631690979, "rewards/total_composite/std": 0.18807001411914825, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0089141130447388, "sampling/importance_sampling_ratio/min": 0.29962798953056335, "sampling/sampling_logp_difference/max": 1.7198677062988281, "sampling/sampling_logp_difference/mean": 0.06001806631684303, "step": 1147 }, { "clip_ratio/high_max": 0.00657894741743803, "clip_ratio/high_mean": 0.00657894741743803, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.00657894741743803, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 37.625, "completions/mean_terminated_length": 37.625, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.03891826095059514, "epoch": 0.046109973089127204, "frac_reward_zero_std": 0.0, "grad_norm": 5.76182746887207, "learning_rate": 6.524242424242425e-06, "loss": 0.0136, "num_tokens": 2586905.0, "reward": 0.9881135821342468, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9881135821342468, "reward_meter_std": 0.0235806442797184, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02358064614236355, "reward_total_composite_mean": 0.9881135821342468, "reward_total_composite_std": 0.0235806442797184, "reward_total_mean": 0.9881135821342468, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9881135821342468, "rewards/meter/std": 0.0235806442797184, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9881135821342468, "rewards/total_composite/std": 0.0235806442797184, "sampling/importance_sampling_ratio/max": 1.4336398839950562, "sampling/importance_sampling_ratio/mean": 1.0021705627441406, "sampling/importance_sampling_ratio/min": 0.6159754991531372, "sampling/sampling_logp_difference/max": 0.48454809188842773, "sampling/sampling_logp_difference/mean": 0.010363386012613773, "step": 1148 }, { "clip_ratio/high_max": 0.021323838154785335, "clip_ratio/high_mean": 0.021323838154785335, "clip_ratio/low_mean": 0.011157091706991196, "clip_ratio/low_min": 0.011157091706991196, "clip_ratio/region_mean": 0.03248092986177653, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 77.0, "completions/mean_terminated_length": 77.0, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.11253493744879961, "epoch": 0.04615013857091216, "frac_reward_zero_std": 0.0, "grad_norm": 2.983004331588745, "learning_rate": 6.521212121212121e-06, "loss": 0.0133, "num_tokens": 2588889.0, "reward": 0.9116114377975464, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948194026947021, "reward_meter_std": 0.007181616500020027, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.1519550234079361, "reward_total_composite_mean": 0.9116114377975464, "reward_total_composite_std": 0.1519550234079361, "reward_total_mean": 0.9116114377975464, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948194026947021, "rewards/meter/std": 0.007181616500020027, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.9116114377975464, "rewards/total_composite/std": 0.1519550234079361, "sampling/importance_sampling_ratio/max": 1.373502492904663, "sampling/importance_sampling_ratio/mean": 1.003957748413086, "sampling/importance_sampling_ratio/min": 0.4784848093986511, "sampling/sampling_logp_difference/max": 0.7371308207511902, "sampling/sampling_logp_difference/mean": 0.022046150639653206, "step": 1149 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.000589622650295496, "clip_ratio/low_min": 0.000589622650295496, "clip_ratio/region_mean": 0.000589622650295496, "completions/clipped_ratio": 0.0, "completions/max_length": 218.0, "completions/max_terminated_length": 218.0, "completions/mean_length": 213.375, "completions/mean_terminated_length": 213.375, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.006337960890959948, "epoch": 0.04619030405269711, "frac_reward_zero_std": 0.0, "grad_norm": 1.0385583639144897, "learning_rate": 6.5181818181818195e-06, "loss": -0.0071, "num_tokens": 2592268.0, "reward": 0.5073539018630981, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996587872505188, "reward_meter_std": 0.0006334602949209511, "reward_repeat_penalty_mean": 0.6363636255264282, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00032248429488390684, "reward_total_composite_mean": 0.5073539018630981, "reward_total_composite_std": 0.0003224806860089302, "reward_total_mean": 0.5073539018630981, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996587872505188, "rewards/meter/std": 0.0006334602949209511, "rewards/repeat_penalty/mean": 0.6363636255264282, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5073539018630981, "rewards/total_composite/std": 0.0003224806860089302, "sampling/importance_sampling_ratio/max": 1.7144917249679565, "sampling/importance_sampling_ratio/mean": 1.000615119934082, "sampling/importance_sampling_ratio/min": 0.5140725374221802, "sampling/sampling_logp_difference/max": 0.6653909683227539, "sampling/sampling_logp_difference/mean": 0.0020132821518927813, "step": 1150 }, { "epoch": 0.04619030405269711, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 363.6923076923077, "eval_completions/max_terminated_length": 363.6923076923077, "eval_completions/mean_length": 209.30769230769232, "eval_completions/mean_terminated_length": 209.30769230769232, "eval_completions/min_length": 63.53846153846154, "eval_completions/min_terminated_length": 63.53846153846154, "eval_entropy": 0.06449572856609638, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 2592268.0, "eval_reward": 0.39963403802651626, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.8692530210201557, "eval_reward_count_adherence_std": 0.11666750907897949, "eval_reward_meter_mean": 0.6754777202239404, "eval_reward_meter_std": 0.40071778343274045, "eval_reward_repeat_penalty_mean": 0.6637609280072726, "eval_reward_repeat_penalty_std": 0.22615024103568152, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.39963403802651626, "eval_reward_total_composite_std": 0.3056422472000122, "eval_reward_total_mean": 0.39963403802651626, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.8692530210201557, "eval_rewards/count_adherence/std": 0.11666750907897949, "eval_rewards/meter/mean": 0.6754777202239404, "eval_rewards/meter/std": 0.40071778343274045, "eval_rewards/repeat_penalty/mean": 0.6637609280072726, "eval_rewards/repeat_penalty/std": 0.22615024103568152, "eval_rewards/total_composite/mean": 0.39963403802651626, "eval_rewards/total_composite/std": 0.3056422472000122, "eval_runtime": 69.6474, "eval_samples_per_second": 1.493, "eval_sampling/importance_sampling_ratio/max": 1.3572269219618578, "eval_sampling/importance_sampling_ratio/mean": 1.001767112658574, "eval_sampling/importance_sampling_ratio/min": 0.4256615604345615, "eval_sampling/sampling_logp_difference/max": 0.8880168061990005, "eval_sampling/sampling_logp_difference/mean": 0.007615503783409412, "eval_steps_per_second": 0.187, "step": 1150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0007062146905809641, "clip_ratio/low_min": 0.0007062146905809641, "clip_ratio/region_mean": 0.0007062146905809641, "completions/clipped_ratio": 0.0, "completions/max_length": 181.0, "completions/max_terminated_length": 181.0, "completions/mean_length": 180.5, "completions/mean_terminated_length": 180.5, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.0079670658451505, "epoch": 0.046230469534482066, "frac_reward_zero_std": 0.0, "grad_norm": 0.24846500158309937, "learning_rate": 6.515151515151516e-06, "loss": -0.0045, "num_tokens": 2595104.0, "reward": 0.4987540543079376, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975081086158752, "reward_meter_std": 0.000507785240188241, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002538926200941205, "reward_total_composite_mean": 0.4987540543079376, "reward_total_composite_std": 0.0002538926200941205, "reward_total_mean": 0.4987540543079376, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975081086158752, "rewards/meter/std": 0.000507785240188241, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4987540543079376, "rewards/total_composite/std": 0.0002538926200941205, "sampling/importance_sampling_ratio/max": 1.6862051486968994, "sampling/importance_sampling_ratio/mean": 1.0008914470672607, "sampling/importance_sampling_ratio/min": 0.90339595079422, "sampling/sampling_logp_difference/max": 0.5224804878234863, "sampling/sampling_logp_difference/mean": 0.0012201471254229546, "step": 1151 }, { "clip_ratio/high_max": 0.03683905629441142, "clip_ratio/high_mean": 0.03683905629441142, "clip_ratio/low_mean": 0.020571128465235233, "clip_ratio/low_min": 0.020571128465235233, "clip_ratio/region_mean": 0.057410184759646654, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.3189387395977974, "epoch": 0.04627063501626702, "frac_reward_zero_std": 0.0, "grad_norm": 17.063459396362305, "learning_rate": 6.512121212121213e-06, "loss": 0.0463, "num_tokens": 2596884.0, "reward": 0.8263880014419556, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8263880014419556, "reward_meter_std": 0.2611541450023651, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2611541450023651, "reward_total_composite_mean": 0.8263880014419556, "reward_total_composite_std": 0.2611541450023651, "reward_total_mean": 0.8263880014419556, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8263880014419556, "rewards/meter/std": 0.2611541450023651, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8263880014419556, "rewards/total_composite/std": 0.2611541450023651, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0026739835739136, "sampling/importance_sampling_ratio/min": 0.012377207167446613, "sampling/sampling_logp_difference/max": 4.3918986320495605, "sampling/sampling_logp_difference/mean": 0.06748796254396439, "step": 1152 }, { "clip_ratio/high_max": 0.0021929824724793434, "clip_ratio/high_mean": 0.0021929824724793434, "clip_ratio/low_mean": 0.018382577574811876, "clip_ratio/low_min": 0.018382577574811876, "clip_ratio/region_mean": 0.02057556004729122, "completions/clipped_ratio": 0.0, "completions/max_length": 119.0, "completions/max_terminated_length": 119.0, "completions/mean_length": 114.875, "completions/mean_terminated_length": 114.875, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.1615551235154271, "epoch": 0.046310800498051974, "frac_reward_zero_std": 0.0, "grad_norm": 3.4035446643829346, "learning_rate": 6.5090909090909095e-06, "loss": 0.0129, "num_tokens": 2599043.0, "reward": 0.8219736814498901, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962472319602966, "reward_meter_std": 0.0029050002340227365, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07157646119594574, "reward_total_composite_mean": 0.8219736814498901, "reward_total_composite_std": 0.07157647609710693, "reward_total_mean": 0.8219736814498901, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962472319602966, "rewards/meter/std": 0.0029050002340227365, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8219736814498901, "rewards/total_composite/std": 0.07157647609710693, "sampling/importance_sampling_ratio/max": 1.9995946884155273, "sampling/importance_sampling_ratio/mean": 0.9999147057533264, "sampling/importance_sampling_ratio/min": 0.283601850271225, "sampling/sampling_logp_difference/max": 1.2601839303970337, "sampling/sampling_logp_difference/mean": 0.0268792062997818, "step": 1153 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 37.0, "completions/mean_terminated_length": 37.0, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.014172183815389872, "epoch": 0.04635096597983693, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.506060606060607e-06, "loss": 0.0, "num_tokens": 2600507.0, "reward": 0.9963316321372986, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963316321372986, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9963316321372986, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9963316321372986, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963316321372986, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963316321372986, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0367541313171387, "sampling/importance_sampling_ratio/mean": 1.0005978345870972, "sampling/importance_sampling_ratio/min": 0.9514572024345398, "sampling/sampling_logp_difference/max": 0.04976058006286621, "sampling/sampling_logp_difference/mean": 0.0014280767645686865, "step": 1154 }, { "clip_ratio/high_max": 0.009366041573230177, "clip_ratio/high_mean": 0.009366041573230177, "clip_ratio/low_mean": 0.005830406676977873, "clip_ratio/low_min": 0.005830406676977873, "clip_ratio/region_mean": 0.01519644825020805, "completions/clipped_ratio": 0.0, "completions/max_length": 241.0, "completions/max_terminated_length": 241.0, "completions/mean_length": 222.625, "completions/mean_terminated_length": 222.625, "completions/min_length": 213.0, "completions/min_terminated_length": 213.0, "entropy": 0.04108358500525355, "epoch": 0.04639113146162188, "frac_reward_zero_std": 0.0, "grad_norm": 1.2776048183441162, "learning_rate": 6.503030303030303e-06, "loss": -0.0352, "num_tokens": 2603752.0, "reward": 0.2860187888145447, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973673820495605, "reward_meter_std": 0.000317126396112144, "reward_repeat_penalty_mean": 0.3440934121608734, "reward_repeat_penalty_std": 0.21156011521816254, "reward_std": 0.17588701844215393, "reward_total_composite_mean": 0.2860187888145447, "reward_total_composite_std": 0.17588701844215393, "reward_total_mean": 0.2860187888145447, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973673820495605, "rewards/meter/std": 0.000317126396112144, "rewards/repeat_penalty/mean": 0.3440934121608734, "rewards/repeat_penalty/std": 0.21156011521816254, "rewards/total_composite/mean": 0.2860187888145447, "rewards/total_composite/std": 0.17588701844215393, "sampling/importance_sampling_ratio/max": 1.5816864967346191, "sampling/importance_sampling_ratio/mean": 0.9986491203308105, "sampling/importance_sampling_ratio/min": 0.11587633937597275, "sampling/sampling_logp_difference/max": 2.1552317142486572, "sampling/sampling_logp_difference/mean": 0.012430640868842602, "step": 1155 }, { "clip_ratio/high_max": 0.026069519110023975, "clip_ratio/high_mean": 0.026069519110023975, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.02974598971195519, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 33.5, "completions/mean_terminated_length": 33.5, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.16383972112089396, "epoch": 0.046431296943406836, "frac_reward_zero_std": 0.0, "grad_norm": 7.644164562225342, "learning_rate": 6.5000000000000004e-06, "loss": 0.0178, "num_tokens": 2605292.0, "reward": 0.9726088643074036, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9726088643074036, "reward_meter_std": 0.043839771300554276, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04383978247642517, "reward_total_composite_mean": 0.9726088643074036, "reward_total_composite_std": 0.043839771300554276, "reward_total_mean": 0.9726088643074036, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9726088643074036, "rewards/meter/std": 0.043839771300554276, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9726088643074036, "rewards/total_composite/std": 0.043839771300554276, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0058238506317139, "sampling/importance_sampling_ratio/min": 0.18433360755443573, "sampling/sampling_logp_difference/max": 1.6910080909729004, "sampling/sampling_logp_difference/mean": 0.041076283901929855, "step": 1156 }, { "clip_ratio/high_max": 0.004000256070867181, "clip_ratio/high_mean": 0.004000256070867181, "clip_ratio/low_mean": 0.03776956582441926, "clip_ratio/low_min": 0.03776956582441926, "clip_ratio/region_mean": 0.04176982189528644, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.75, "completions/mean_terminated_length": 62.75, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.1482592076063156, "epoch": 0.04647146242519179, "frac_reward_zero_std": 0.0, "grad_norm": 5.127370357513428, "learning_rate": 6.496969696969697e-06, "loss": 0.0036, "num_tokens": 2607186.0, "reward": 0.4182388186454773, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4182388186454773, "reward_meter_std": 0.25260862708091736, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.25260859727859497, "reward_total_composite_mean": 0.4182388186454773, "reward_total_composite_std": 0.25260862708091736, "reward_total_mean": 0.4182388186454773, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4182388186454773, "rewards/meter/std": 0.25260862708091736, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4182388186454773, "rewards/total_composite/std": 0.25260862708091736, "sampling/importance_sampling_ratio/max": 1.7306538820266724, "sampling/importance_sampling_ratio/mean": 1.0079007148742676, "sampling/importance_sampling_ratio/min": 0.2224455326795578, "sampling/sampling_logp_difference/max": 1.50307297706604, "sampling/sampling_logp_difference/mean": 0.03417876362800598, "step": 1157 }, { "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/low_mean": 0.008780332282185555, "clip_ratio/low_min": 0.008780332282185555, "clip_ratio/region_mean": 0.01332578668370843, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 55.25, "completions/mean_terminated_length": 55.25, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.07266200426965952, "epoch": 0.046511627906976744, "frac_reward_zero_std": 0.0, "grad_norm": 3.4404122829437256, "learning_rate": 6.493939393939395e-06, "loss": 0.0042, "num_tokens": 2608844.0, "reward": 0.09517530351877213, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.1349601447582245, "reward_meter_std": 0.012392125092446804, "reward_repeat_penalty_mean": 0.7083333730697632, "reward_repeat_penalty_std": 0.11785111576318741, "reward_std": 0.01430203951895237, "reward_total_composite_mean": 0.09517530351877213, "reward_total_composite_std": 0.01430203951895237, "reward_total_mean": 0.09517530351877213, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.1349601447582245, "rewards/meter/std": 0.012392125092446804, "rewards/repeat_penalty/mean": 0.7083333730697632, "rewards/repeat_penalty/std": 0.11785111576318741, "rewards/total_composite/mean": 0.09517530351877213, "rewards/total_composite/std": 0.01430203951895237, "sampling/importance_sampling_ratio/max": 1.3683043718338013, "sampling/importance_sampling_ratio/mean": 1.0016802549362183, "sampling/importance_sampling_ratio/min": 0.2998819947242737, "sampling/sampling_logp_difference/max": 1.2043662071228027, "sampling/sampling_logp_difference/mean": 0.016672344878315926, "step": 1158 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.011482007801532745, "clip_ratio/low_min": 0.011482007801532745, "clip_ratio/region_mean": 0.015269886702299118, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.75, "completions/mean_terminated_length": 32.75, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0744382287375629, "epoch": 0.0465517933887617, "frac_reward_zero_std": 0.0, "grad_norm": 10.866904258728027, "learning_rate": 6.490909090909091e-06, "loss": 0.0034, "num_tokens": 2610338.0, "reward": 0.9911776781082153, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9911776781082153, "reward_meter_std": 0.0027705347165465355, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002770546358078718, "reward_total_composite_mean": 0.9911776781082153, "reward_total_composite_std": 0.0027705347165465355, "reward_total_mean": 0.9911776781082153, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9911776781082153, "rewards/meter/std": 0.0027705347165465355, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9911776781082153, "rewards/total_composite/std": 0.0027705347165465355, "sampling/importance_sampling_ratio/max": 1.708295226097107, "sampling/importance_sampling_ratio/mean": 1.002776861190796, "sampling/importance_sampling_ratio/min": 0.3791012763977051, "sampling/sampling_logp_difference/max": 0.969951868057251, "sampling/sampling_logp_difference/mean": 0.017460720613598824, "step": 1159 }, { "clip_ratio/high_max": 0.0021929824724793434, "clip_ratio/high_mean": 0.0021929824724793434, "clip_ratio/low_mean": 0.004551473073661327, "clip_ratio/low_min": 0.004551473073661327, "clip_ratio/region_mean": 0.006744455546140671, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 56.5, "completions/mean_terminated_length": 56.5, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.07584757776930928, "epoch": 0.04659195887054665, "frac_reward_zero_std": 0.0, "grad_norm": 5.433602809906006, "learning_rate": 6.487878787878789e-06, "loss": 0.0007, "num_tokens": 2612174.0, "reward": 0.19516333937644958, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.19516333937644958, "reward_meter_std": 0.10145802795886993, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.10145802050828934, "reward_total_composite_mean": 0.19516333937644958, "reward_total_composite_std": 0.10145802795886993, "reward_total_mean": 0.19516333937644958, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.19516333937644958, "rewards/meter/std": 0.10145802795886993, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.19516333937644958, "rewards/total_composite/std": 0.10145802795886993, "sampling/importance_sampling_ratio/max": 1.9637987613677979, "sampling/importance_sampling_ratio/mean": 1.0025193691253662, "sampling/importance_sampling_ratio/min": 0.009674739092588425, "sampling/sampling_logp_difference/max": 4.638236999511719, "sampling/sampling_logp_difference/mean": 0.030529655516147614, "step": 1160 }, { "clip_ratio/high_max": 0.011600378900766373, "clip_ratio/high_mean": 0.011600378900766373, "clip_ratio/low_mean": 0.02016128972172737, "clip_ratio/low_min": 0.02016128972172737, "clip_ratio/region_mean": 0.031761668622493744, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 31.875, "completions/mean_terminated_length": 31.875, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.11351501289755106, "epoch": 0.046632124352331605, "frac_reward_zero_std": 0.0, "grad_norm": 12.443059921264648, "learning_rate": 6.484848484848485e-06, "loss": -0.0148, "num_tokens": 2613517.0, "reward": 0.9917822480201721, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9917822480201721, "reward_meter_std": 0.003353093983605504, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003353102831169963, "reward_total_composite_mean": 0.9917822480201721, "reward_total_composite_std": 0.003353093983605504, "reward_total_mean": 0.9917822480201721, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9917822480201721, "rewards/meter/std": 0.003353093983605504, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917822480201721, "rewards/total_composite/std": 0.003353093983605504, "sampling/importance_sampling_ratio/max": 1.7660976648330688, "sampling/importance_sampling_ratio/mean": 1.0028786659240723, "sampling/importance_sampling_ratio/min": 0.6050869226455688, "sampling/sampling_logp_difference/max": 0.5687723755836487, "sampling/sampling_logp_difference/mean": 0.02346481755375862, "step": 1161 }, { "clip_ratio/high_max": 0.027328122640028596, "clip_ratio/high_mean": 0.027328122640028596, "clip_ratio/low_mean": 0.005444317124783993, "clip_ratio/low_min": 0.005444317124783993, "clip_ratio/region_mean": 0.03277243976481259, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 116.25, "completions/mean_terminated_length": 116.25, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.16215351317077875, "epoch": 0.04667228983411656, "frac_reward_zero_std": 0.0, "grad_norm": 3.7559328079223633, "learning_rate": 6.481818181818182e-06, "loss": 0.0034, "num_tokens": 2615967.0, "reward": 0.891869068145752, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9908610582351685, "reward_meter_std": 0.010183922946453094, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.10717973858118057, "reward_total_composite_mean": 0.891869068145752, "reward_total_composite_std": 0.10717974603176117, "reward_total_mean": 0.891869068145752, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9908610582351685, "rewards/meter/std": 0.010183922946453094, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.891869068145752, "rewards/total_composite/std": 0.10717974603176117, "sampling/importance_sampling_ratio/max": 1.661534309387207, "sampling/importance_sampling_ratio/mean": 0.9961613416671753, "sampling/importance_sampling_ratio/min": 0.17334923148155212, "sampling/sampling_logp_difference/max": 1.7524470090866089, "sampling/sampling_logp_difference/mean": 0.03219734504818916, "step": 1162 }, { "clip_ratio/high_max": 0.003341847565025091, "clip_ratio/high_mean": 0.003341847565025091, "clip_ratio/low_mean": 0.003117707441560924, "clip_ratio/low_min": 0.003117707441560924, "clip_ratio/region_mean": 0.006459555006586015, "completions/clipped_ratio": 0.0, "completions/max_length": 232.0, "completions/max_terminated_length": 232.0, "completions/mean_length": 211.375, "completions/mean_terminated_length": 211.375, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "entropy": 0.044531215680763125, "epoch": 0.04671245531590151, "frac_reward_zero_std": 0.0, "grad_norm": 2.036916971206665, "learning_rate": 6.478787878787879e-06, "loss": -0.0674, "num_tokens": 2619154.0, "reward": 0.31233876943588257, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8547403812408447, "reward_meter_std": 0.34573426842689514, "reward_repeat_penalty_mean": 0.4711538553237915, "reward_repeat_penalty_std": 0.2268439084291458, "reward_std": 0.20915129780769348, "reward_total_composite_mean": 0.31233876943588257, "reward_total_composite_std": 0.20915131270885468, "reward_total_mean": 0.31233876943588257, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8547403812408447, "rewards/meter/std": 0.34573426842689514, "rewards/repeat_penalty/mean": 0.4711538553237915, "rewards/repeat_penalty/std": 0.2268439084291458, "rewards/total_composite/mean": 0.31233876943588257, "rewards/total_composite/std": 0.20915131270885468, "sampling/importance_sampling_ratio/max": 1.551490068435669, "sampling/importance_sampling_ratio/mean": 0.9999427795410156, "sampling/importance_sampling_ratio/min": 0.41830188035964966, "sampling/sampling_logp_difference/max": 0.8715519905090332, "sampling/sampling_logp_difference/mean": 0.010163981467485428, "step": 1163 }, { "clip_ratio/high_max": 0.014912280952557921, "clip_ratio/high_mean": 0.014912280952557921, "clip_ratio/low_mean": 0.0016447368543595076, "clip_ratio/low_min": 0.0016447368543595076, "clip_ratio/region_mean": 0.01655701780691743, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 75.5, "completions/mean_terminated_length": 75.5, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.06199400080367923, "epoch": 0.04675262079768647, "frac_reward_zero_std": 0.0, "grad_norm": 4.857386112213135, "learning_rate": 6.475757575757576e-06, "loss": 0.0062, "num_tokens": 2621022.0, "reward": 0.905701756477356, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.905701756477356, "reward_meter_std": 0.22805918753147125, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22805917263031006, "reward_total_composite_mean": 0.905701756477356, "reward_total_composite_std": 0.22805918753147125, "reward_total_mean": 0.905701756477356, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.905701756477356, "rewards/meter/std": 0.22805918753147125, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.905701756477356, "rewards/total_composite/std": 0.22805918753147125, "sampling/importance_sampling_ratio/max": 1.5772894620895386, "sampling/importance_sampling_ratio/mean": 1.003965139389038, "sampling/importance_sampling_ratio/min": 0.38787469267845154, "sampling/sampling_logp_difference/max": 0.9470729827880859, "sampling/sampling_logp_difference/mean": 0.014857825823128223, "step": 1164 }, { "clip_ratio/high_max": 0.02869208576157689, "clip_ratio/high_mean": 0.02869208576157689, "clip_ratio/low_mean": 0.009149970952421427, "clip_ratio/low_min": 0.009149970952421427, "clip_ratio/region_mean": 0.03784205671399832, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 40.25, "completions/mean_terminated_length": 40.25, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.19564981944859028, "epoch": 0.04679278627947142, "frac_reward_zero_std": 0.0, "grad_norm": 4.955336570739746, "learning_rate": 6.472727272727272e-06, "loss": 0.0288, "num_tokens": 2622496.0, "reward": 0.9979323148727417, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979323148727417, "reward_meter_std": 0.0005710619152523577, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005710580153390765, "reward_total_composite_mean": 0.9979323148727417, "reward_total_composite_std": 0.0005710619152523577, "reward_total_mean": 0.9979323148727417, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979323148727417, "rewards/meter/std": 0.0005710619152523577, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979323148727417, "rewards/total_composite/std": 0.0005710619152523577, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9936442971229553, "sampling/importance_sampling_ratio/min": 0.3421395719051361, "sampling/sampling_logp_difference/max": 1.0725364685058594, "sampling/sampling_logp_difference/mean": 0.05128004774451256, "step": 1165 }, { "clip_ratio/high_max": 0.014788191299885511, "clip_ratio/high_mean": 0.014788191299885511, "clip_ratio/low_mean": 0.007695082924328744, "clip_ratio/low_min": 0.007695082924328744, "clip_ratio/region_mean": 0.022483274224214256, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.0714933155104518, "epoch": 0.046832951761256375, "frac_reward_zero_std": 0.0, "grad_norm": 10.79326057434082, "learning_rate": 6.4696969696969705e-06, "loss": -0.0054, "num_tokens": 2624300.0, "reward": 0.9930996894836426, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9930996894836426, "reward_meter_std": 0.0033798664808273315, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003379874862730503, "reward_total_composite_mean": 0.9930996894836426, "reward_total_composite_std": 0.0033798664808273315, "reward_total_mean": 0.9930996894836426, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9930996894836426, "rewards/meter/std": 0.0033798664808273315, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9930996894836426, "rewards/total_composite/std": 0.0033798664808273315, "sampling/importance_sampling_ratio/max": 1.7809786796569824, "sampling/importance_sampling_ratio/mean": 0.9978261590003967, "sampling/importance_sampling_ratio/min": 0.08999272435903549, "sampling/sampling_logp_difference/max": 2.4080264568328857, "sampling/sampling_logp_difference/mean": 0.02763393148779869, "step": 1166 }, { "clip_ratio/high_max": 0.0012690355069935322, "clip_ratio/high_mean": 0.0012690355069935322, "clip_ratio/low_mean": 0.005486392183229327, "clip_ratio/low_min": 0.005486392183229327, "clip_ratio/region_mean": 0.006755427690222859, "completions/clipped_ratio": 0.0, "completions/max_length": 210.0, "completions/max_terminated_length": 210.0, "completions/mean_length": 203.0, "completions/mean_terminated_length": 203.0, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "entropy": 0.020221689250320196, "epoch": 0.04687311724304133, "frac_reward_zero_std": 0.0, "grad_norm": 0.7076108455657959, "learning_rate": 6.466666666666667e-06, "loss": 0.0106, "num_tokens": 2627412.0, "reward": 0.03625128045678139, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.07068999111652374, "reward_meter_std": 0.12723270058631897, "reward_repeat_penalty_mean": 0.6153846383094788, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06524752825498581, "reward_total_composite_mean": 0.03625128045678139, "reward_total_composite_std": 0.0652475357055664, "reward_total_mean": 0.03625128045678139, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.07068999111652374, "rewards/meter/std": 0.12723270058631897, "rewards/repeat_penalty/mean": 0.6153846383094788, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.03625128045678139, "rewards/total_composite/std": 0.0652475357055664, "sampling/importance_sampling_ratio/max": 1.9165208339691162, "sampling/importance_sampling_ratio/mean": 1.00026535987854, "sampling/importance_sampling_ratio/min": 0.12294920533895493, "sampling/sampling_logp_difference/max": 2.0959839820861816, "sampling/sampling_logp_difference/mean": 0.005202981643378735, "step": 1167 }, { "clip_ratio/high_max": 0.00147058826405555, "clip_ratio/high_mean": 0.00147058826405555, "clip_ratio/low_mean": 0.0043441514717414975, "clip_ratio/low_min": 0.0043441514717414975, "clip_ratio/region_mean": 0.005814739735797048, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 86.375, "completions/mean_terminated_length": 86.375, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.04172563157044351, "epoch": 0.04691328272482628, "frac_reward_zero_std": 0.0, "grad_norm": 3.631923198699951, "learning_rate": 6.463636363636364e-06, "loss": 0.0086, "num_tokens": 2629391.0, "reward": 0.06269732117652893, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.07566258311271667, "reward_meter_std": 0.10104013234376907, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.08014770597219467, "reward_total_composite_mean": 0.06269732117652893, "reward_total_composite_std": 0.08014771342277527, "reward_total_mean": 0.06269732117652893, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.07566258311271667, "rewards/meter/std": 0.10104013234376907, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.06269732117652893, "rewards/total_composite/std": 0.08014771342277527, "sampling/importance_sampling_ratio/max": 1.388146162033081, "sampling/importance_sampling_ratio/mean": 0.9974919557571411, "sampling/importance_sampling_ratio/min": 0.35537731647491455, "sampling/sampling_logp_difference/max": 1.0345752239227295, "sampling/sampling_logp_difference/mean": 0.009703797288239002, "step": 1168 }, { "clip_ratio/high_max": 0.00317868051934056, "clip_ratio/high_mean": 0.00317868051934056, "clip_ratio/low_mean": 0.0029717457364313304, "clip_ratio/low_min": 0.0029717457364313304, "clip_ratio/region_mean": 0.00615042625577189, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 262.25, "completions/mean_terminated_length": 262.25, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.047941830940544605, "epoch": 0.04695344820661124, "frac_reward_zero_std": 0.0, "grad_norm": 1.607576847076416, "learning_rate": 6.460606060606061e-06, "loss": -0.0418, "num_tokens": 2633025.0, "reward": 0.5507833957672119, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962061643600464, "reward_meter_std": 0.0021561330650001764, "reward_repeat_penalty_mean": 0.6634615659713745, "reward_repeat_penalty_std": 0.05723259598016739, "reward_std": 0.04750329256057739, "reward_total_composite_mean": 0.5507833957672119, "reward_total_composite_std": 0.04750329256057739, "reward_total_mean": 0.5507833957672119, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962061643600464, "rewards/meter/std": 0.0021561330650001764, "rewards/repeat_penalty/mean": 0.6634615659713745, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.5507833957672119, "rewards/total_composite/std": 0.04750329256057739, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0013923645019531, "sampling/importance_sampling_ratio/min": 0.26845112442970276, "sampling/sampling_logp_difference/max": 1.3150863647460938, "sampling/sampling_logp_difference/mean": 0.010979145765304565, "step": 1169 }, { "clip_ratio/high_max": 0.02818627143278718, "clip_ratio/high_mean": 0.02818627143278718, "clip_ratio/low_mean": 0.0031645570416003466, "clip_ratio/low_min": 0.0031645570416003466, "clip_ratio/region_mean": 0.031350828474387527, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.24884118884801865, "epoch": 0.04699361368839619, "frac_reward_zero_std": 0.0, "grad_norm": 4.66847562789917, "learning_rate": 6.457575757575758e-06, "loss": -0.0059, "num_tokens": 2635041.0, "reward": 0.9650785326957703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9650785326957703, "reward_meter_std": 0.06653696298599243, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06653697043657303, "reward_total_composite_mean": 0.9650785326957703, "reward_total_composite_std": 0.06653696298599243, "reward_total_mean": 0.9650785326957703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9650785326957703, "rewards/meter/std": 0.06653696298599243, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9650785326957703, "rewards/total_composite/std": 0.06653696298599243, "sampling/importance_sampling_ratio/max": 1.8339917659759521, "sampling/importance_sampling_ratio/mean": 1.0022578239440918, "sampling/importance_sampling_ratio/min": 0.2196621596813202, "sampling/sampling_logp_difference/max": 1.5156645774841309, "sampling/sampling_logp_difference/mean": 0.03951912745833397, "step": 1170 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 85.0, "completions/mean_terminated_length": 85.0, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.023862487636506557, "epoch": 0.047033779170181145, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.454545454545456e-06, "loss": 0.0, "num_tokens": 2636953.0, "reward": 0.2603916525840759, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3254895508289337, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.2603916525840759, "reward_total_composite_std": 0.0, "reward_total_mean": 0.2603916525840759, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3254895508289337, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2603916525840759, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.1062525510787964, "sampling/importance_sampling_ratio/mean": 1.0015860795974731, "sampling/importance_sampling_ratio/min": 0.9174838662147522, "sampling/sampling_logp_difference/max": 0.10097821056842804, "sampling/sampling_logp_difference/mean": 0.0026430224534124136, "step": 1171 }, { "clip_ratio/high_max": 0.015921411802992225, "clip_ratio/high_mean": 0.015921411802992225, "clip_ratio/low_mean": 0.010513697401620448, "clip_ratio/low_min": 0.010513697401620448, "clip_ratio/region_mean": 0.026435109204612672, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 118.125, "completions/mean_terminated_length": 118.125, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.13357019051909447, "epoch": 0.0470739446519661, "frac_reward_zero_std": 0.0, "grad_norm": 4.03551721572876, "learning_rate": 6.451515151515152e-06, "loss": 0.008, "num_tokens": 2639474.0, "reward": 0.921150803565979, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958237409591675, "reward_meter_std": 0.0014982123393565416, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10321993380784988, "reward_total_composite_mean": 0.921150803565979, "reward_total_composite_std": 0.10321994870901108, "reward_total_mean": 0.921150803565979, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958237409591675, "rewards/meter/std": 0.0014982123393565416, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.921150803565979, "rewards/total_composite/std": 0.10321994870901108, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001880407333374, "sampling/importance_sampling_ratio/min": 0.2408369481563568, "sampling/sampling_logp_difference/max": 1.4236351251602173, "sampling/sampling_logp_difference/mean": 0.027086030691862106, "step": 1172 }, { "clip_ratio/high_max": 0.008183791185729206, "clip_ratio/high_mean": 0.008183791185729206, "clip_ratio/low_mean": 0.004870129749178886, "clip_ratio/low_min": 0.004870129749178886, "clip_ratio/region_mean": 0.013053920934908092, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.375, "completions/mean_terminated_length": 76.375, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.08165451744571328, "epoch": 0.04711411013375105, "frac_reward_zero_std": 0.0, "grad_norm": 3.1960055828094482, "learning_rate": 6.4484848484848496e-06, "loss": 0.008, "num_tokens": 2641397.0, "reward": 0.988898754119873, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.988898754119873, "reward_meter_std": 0.0167799461632967, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.016779951751232147, "reward_total_composite_mean": 0.988898754119873, "reward_total_composite_std": 0.0167799461632967, "reward_total_mean": 0.988898754119873, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.988898754119873, "rewards/meter/std": 0.0167799461632967, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.988898754119873, "rewards/total_composite/std": 0.0167799461632967, "sampling/importance_sampling_ratio/max": 1.5719304084777832, "sampling/importance_sampling_ratio/mean": 1.0082311630249023, "sampling/importance_sampling_ratio/min": 0.3004509508609772, "sampling/sampling_logp_difference/max": 1.2024707794189453, "sampling/sampling_logp_difference/mean": 0.015661096200346947, "step": 1173 }, { "clip_ratio/high_max": 0.003289473708719015, "clip_ratio/high_mean": 0.003289473708719015, "clip_ratio/low_mean": 0.0033333334140479565, "clip_ratio/low_min": 0.0033333334140479565, "clip_ratio/region_mean": 0.006622807122766972, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 75.875, "completions/mean_terminated_length": 75.875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.08360559307038784, "epoch": 0.04715427561553601, "frac_reward_zero_std": 0.0, "grad_norm": 3.3341424465179443, "learning_rate": 6.445454545454546e-06, "loss": -0.0038, "num_tokens": 2643116.0, "reward": 0.9900845289230347, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9900845289230347, "reward_meter_std": 0.006058650091290474, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006058644037693739, "reward_total_composite_mean": 0.9900845289230347, "reward_total_composite_std": 0.006058650091290474, "reward_total_mean": 0.9900845289230347, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9900845289230347, "rewards/meter/std": 0.006058650091290474, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9900845289230347, "rewards/total_composite/std": 0.006058650091290474, "sampling/importance_sampling_ratio/max": 1.7328146696090698, "sampling/importance_sampling_ratio/mean": 1.0023661851882935, "sampling/importance_sampling_ratio/min": 0.38840532302856445, "sampling/sampling_logp_difference/max": 0.9457058906555176, "sampling/sampling_logp_difference/mean": 0.017089035362005234, "step": 1174 }, { "clip_ratio/high_max": 0.004411764908581972, "clip_ratio/high_mean": 0.004411764908581972, "clip_ratio/low_mean": 0.002794080995954573, "clip_ratio/low_min": 0.002794080995954573, "clip_ratio/region_mean": 0.007205845904536545, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 86.125, "completions/mean_terminated_length": 86.125, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.05733899096958339, "epoch": 0.04719444109732096, "frac_reward_zero_std": 0.0, "grad_norm": 3.098088502883911, "learning_rate": 6.442424242424243e-06, "loss": 0.0211, "num_tokens": 2645197.0, "reward": 0.19161534309387207, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2395191788673401, "reward_meter_std": 0.14119157195091248, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.11295326054096222, "reward_total_composite_mean": 0.19161534309387207, "reward_total_composite_std": 0.11295326054096222, "reward_total_mean": 0.19161534309387207, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2395191788673401, "rewards/meter/std": 0.14119157195091248, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.19161534309387207, "rewards/total_composite/std": 0.11295326054096222, "sampling/importance_sampling_ratio/max": 1.5297051668167114, "sampling/importance_sampling_ratio/mean": 1.0021647214889526, "sampling/importance_sampling_ratio/min": 0.32919374108314514, "sampling/sampling_logp_difference/max": 1.1111087799072266, "sampling/sampling_logp_difference/mean": 0.008148287422955036, "step": 1175 }, { "clip_ratio/high_max": 0.013525328016839921, "clip_ratio/high_mean": 0.013525328016839921, "clip_ratio/low_mean": 0.0026997023087460548, "clip_ratio/low_min": 0.0026997023087460548, "clip_ratio/region_mean": 0.016225030325585976, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 277.125, "completions/mean_terminated_length": 277.125, "completions/min_length": 268.0, "completions/min_terminated_length": 268.0, "entropy": 0.09004434384405613, "epoch": 0.047234606579105914, "frac_reward_zero_std": 0.0, "grad_norm": 1.6563795804977417, "learning_rate": 6.43939393939394e-06, "loss": -0.0027, "num_tokens": 2649214.0, "reward": 0.6114773750305176, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9914854764938354, "reward_meter_std": 0.0076880063861608505, "reward_repeat_penalty_mean": 0.7403846383094788, "reward_repeat_penalty_std": 0.15350237488746643, "reward_std": 0.12588950991630554, "reward_total_composite_mean": 0.6114773750305176, "reward_total_composite_std": 0.12588950991630554, "reward_total_mean": 0.6114773750305176, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9914854764938354, "rewards/meter/std": 0.0076880063861608505, "rewards/repeat_penalty/mean": 0.7403846383094788, "rewards/repeat_penalty/std": 0.15350237488746643, "rewards/total_composite/mean": 0.6114773750305176, "rewards/total_composite/std": 0.12588950991630554, "sampling/importance_sampling_ratio/max": 1.965925931930542, "sampling/importance_sampling_ratio/mean": 1.0004220008850098, "sampling/importance_sampling_ratio/min": 0.3256934881210327, "sampling/sampling_logp_difference/max": 1.1217985153198242, "sampling/sampling_logp_difference/mean": 0.015588824637234211, "step": 1176 }, { "clip_ratio/high_max": 0.00390625, "clip_ratio/high_mean": 0.00390625, "clip_ratio/low_mean": 0.006048386916518211, "clip_ratio/low_min": 0.006048386916518211, "clip_ratio/region_mean": 0.009954636916518211, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 58.5, "completions/mean_terminated_length": 58.5, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.06206447468139231, "epoch": 0.04727477206089087, "frac_reward_zero_std": 0.0, "grad_norm": 4.175478935241699, "learning_rate": 6.436363636363637e-06, "loss": -0.0359, "num_tokens": 2650978.0, "reward": 0.36279141902923584, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.36279141902923584, "reward_meter_std": 0.24407321214675903, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.24407321214675903, "reward_total_composite_mean": 0.36279141902923584, "reward_total_composite_std": 0.24407321214675903, "reward_total_mean": 0.36279141902923584, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.36279141902923584, "rewards/meter/std": 0.24407321214675903, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.36279141902923584, "rewards/total_composite/std": 0.24407321214675903, "sampling/importance_sampling_ratio/max": 1.3822486400604248, "sampling/importance_sampling_ratio/mean": 1.0017210245132446, "sampling/importance_sampling_ratio/min": 0.503181517124176, "sampling/sampling_logp_difference/max": 0.6868042945861816, "sampling/sampling_logp_difference/mean": 0.009452205151319504, "step": 1177 }, { "clip_ratio/high_max": 0.007645888428669423, "clip_ratio/high_mean": 0.007645888428669423, "clip_ratio/low_mean": 0.0021399176912382245, "clip_ratio/low_min": 0.0021399176912382245, "clip_ratio/region_mean": 0.009785806119907647, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 257.0, "completions/mean_length": 267.125, "completions/mean_terminated_length": 232.1428680419922, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.07222116971388459, "epoch": 0.04731493754267582, "frac_reward_zero_std": 0.0, "grad_norm": 1.462255835533142, "learning_rate": 6.433333333333333e-06, "loss": -0.1566, "num_tokens": 2653979.0, "reward": 0.2283807098865509, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.05050762742757797, "reward_meter_mean": 0.4442686140537262, "reward_meter_std": 0.2729472219944, "reward_repeat_penalty_mean": 0.6192708611488342, "reward_repeat_penalty_std": 0.21683759987354279, "reward_std": 0.17121195793151855, "reward_total_composite_mean": 0.2283807098865509, "reward_total_composite_std": 0.17121197283267975, "reward_total_mean": 0.2283807098865509, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.05050762742757797, "rewards/meter/mean": 0.4442686140537262, "rewards/meter/std": 0.2729472219944, "rewards/repeat_penalty/mean": 0.6192708611488342, "rewards/repeat_penalty/std": 0.21683759987354279, "rewards/total_composite/mean": 0.2283807098865509, "rewards/total_composite/std": 0.17121197283267975, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0050114393234253, "sampling/importance_sampling_ratio/min": 0.44583451747894287, "sampling/sampling_logp_difference/max": 1.0156912803649902, "sampling/sampling_logp_difference/mean": 0.012943691574037075, "step": 1178 }, { "clip_ratio/high_max": 0.016267166705802083, "clip_ratio/high_mean": 0.016267166705802083, "clip_ratio/low_mean": 0.014470615307800472, "clip_ratio/low_min": 0.014470615307800472, "clip_ratio/region_mean": 0.030737782013602555, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 68.625, "completions/mean_terminated_length": 68.625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.09837264008820057, "epoch": 0.047355103024460776, "frac_reward_zero_std": 0.0, "grad_norm": 11.408062934875488, "learning_rate": 6.430303030303031e-06, "loss": 0.0133, "num_tokens": 2655696.0, "reward": 0.9946950078010559, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946950078010559, "reward_meter_std": 0.002590329386293888, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002590324031189084, "reward_total_composite_mean": 0.9946950078010559, "reward_total_composite_std": 0.002590329386293888, "reward_total_mean": 0.9946950078010559, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946950078010559, "rewards/meter/std": 0.002590329386293888, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946950078010559, "rewards/total_composite/std": 0.002590329386293888, "sampling/importance_sampling_ratio/max": 1.6397497653961182, "sampling/importance_sampling_ratio/mean": 1.0001428127288818, "sampling/importance_sampling_ratio/min": 0.31258830428123474, "sampling/sampling_logp_difference/max": 1.1628682613372803, "sampling/sampling_logp_difference/mean": 0.026325104758143425, "step": 1179 }, { "clip_ratio/high_max": 0.0242516309954226, "clip_ratio/high_mean": 0.0242516309954226, "clip_ratio/low_mean": 0.011504121124744415, "clip_ratio/low_min": 0.011504121124744415, "clip_ratio/region_mean": 0.03575575212016702, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 55.875, "completions/mean_terminated_length": 55.875, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.142734014429152, "epoch": 0.04739526850624573, "frac_reward_zero_std": 0.0, "grad_norm": 7.856446266174316, "learning_rate": 6.427272727272728e-06, "loss": 0.0028, "num_tokens": 2657383.0, "reward": 0.7185721397399902, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7185721397399902, "reward_meter_std": 0.25754180550575256, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.25754180550575256, "reward_total_composite_mean": 0.7185721397399902, "reward_total_composite_std": 0.25754180550575256, "reward_total_mean": 0.7185721397399902, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7185721397399902, "rewards/meter/std": 0.25754180550575256, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7185721397399902, "rewards/total_composite/std": 0.25754180550575256, "sampling/importance_sampling_ratio/max": 1.7777073383331299, "sampling/importance_sampling_ratio/mean": 0.9931161403656006, "sampling/importance_sampling_ratio/min": 0.2716076672077179, "sampling/sampling_logp_difference/max": 1.3033967018127441, "sampling/sampling_logp_difference/mean": 0.03376821056008339, "step": 1180 }, { "clip_ratio/high_max": 0.006440639495849609, "clip_ratio/high_mean": 0.006440639495849609, "clip_ratio/low_mean": 0.004685206571593881, "clip_ratio/low_min": 0.004685206571593881, "clip_ratio/region_mean": 0.01112584606744349, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 371.375, "completions/mean_terminated_length": 371.375, "completions/min_length": 359.0, "completions/min_terminated_length": 359.0, "entropy": 0.0724606835283339, "epoch": 0.047435433988030684, "frac_reward_zero_std": 0.0, "grad_norm": 1.3380672931671143, "learning_rate": 6.424242424242425e-06, "loss": 0.0075, "num_tokens": 2662002.0, "reward": 0.6446620225906372, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9918642044067383, "reward_meter_std": 0.006125128827989101, "reward_repeat_penalty_mean": 0.7426470518112183, "reward_repeat_penalty_std": 0.08281681686639786, "reward_std": 0.07332959026098251, "reward_total_composite_mean": 0.6446620225906372, "reward_total_composite_std": 0.07332957535982132, "reward_total_mean": 0.6446620225906372, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9918642044067383, "rewards/meter/std": 0.006125128827989101, "rewards/repeat_penalty/mean": 0.7426470518112183, "rewards/repeat_penalty/std": 0.08281681686639786, "rewards/total_composite/mean": 0.6446620225906372, "rewards/total_composite/std": 0.07332957535982132, "sampling/importance_sampling_ratio/max": 1.6264017820358276, "sampling/importance_sampling_ratio/mean": 1.0013055801391602, "sampling/importance_sampling_ratio/min": 0.2434346228837967, "sampling/sampling_logp_difference/max": 1.4129068851470947, "sampling/sampling_logp_difference/mean": 0.013688583858311176, "step": 1181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.008097166079096496, "clip_ratio/low_min": 0.008097166079096496, "clip_ratio/region_mean": 0.008097166079096496, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 77.5, "completions/mean_terminated_length": 77.5, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.04025013535283506, "epoch": 0.04747559946981564, "frac_reward_zero_std": 0.0, "grad_norm": 2.379884719848633, "learning_rate": 6.4212121212121215e-06, "loss": -0.0062, "num_tokens": 2663974.0, "reward": 0.9949516654014587, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949516654014587, "reward_meter_std": 0.0005235640564933419, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005235668504610658, "reward_total_composite_mean": 0.9949516654014587, "reward_total_composite_std": 0.0005235640564933419, "reward_total_mean": 0.9949516654014587, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949516654014587, "rewards/meter/std": 0.0005235640564933419, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949516654014587, "rewards/total_composite/std": 0.0005235640564933419, "sampling/importance_sampling_ratio/max": 1.353779911994934, "sampling/importance_sampling_ratio/mean": 0.9996510148048401, "sampling/importance_sampling_ratio/min": 0.5104632377624512, "sampling/sampling_logp_difference/max": 0.6724367141723633, "sampling/sampling_logp_difference/mean": 0.010198801755905151, "step": 1182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.046004267409443855, "epoch": 0.04751576495160059, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.418181818181819e-06, "loss": 0.0, "num_tokens": 2665278.0, "reward": 0.9889696836471558, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9889696836471558, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9889696836471558, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9889696836471558, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9889696836471558, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9889696836471558, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0598801374435425, "sampling/importance_sampling_ratio/mean": 0.9997879266738892, "sampling/importance_sampling_ratio/min": 0.8542470932006836, "sampling/sampling_logp_difference/max": 0.15753476321697235, "sampling/sampling_logp_difference/mean": 0.005411628168076277, "step": 1183 }, { "clip_ratio/high_max": 0.020877854549326003, "clip_ratio/high_mean": 0.020877854549326003, "clip_ratio/low_mean": 0.012163461884483695, "clip_ratio/low_min": 0.012163461884483695, "clip_ratio/region_mean": 0.0330413164338097, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 97.25, "completions/mean_terminated_length": 97.25, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.16915534622967243, "epoch": 0.047555930433385546, "frac_reward_zero_std": 0.0, "grad_norm": 9.193962097167969, "learning_rate": 6.415151515151515e-06, "loss": 0.038, "num_tokens": 2667336.0, "reward": 0.724637508392334, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.892284095287323, "reward_meter_std": 0.20539914071559906, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.1372780054807663, "reward_total_composite_mean": 0.724637508392334, "reward_total_composite_std": 0.1372780054807663, "reward_total_mean": 0.724637508392334, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.892284095287323, "rewards/meter/std": 0.20539914071559906, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.724637508392334, "rewards/total_composite/std": 0.1372780054807663, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9922058582305908, "sampling/importance_sampling_ratio/min": 0.17699459195137024, "sampling/sampling_logp_difference/max": 1.7316360473632812, "sampling/sampling_logp_difference/mean": 0.04928457364439964, "step": 1184 }, { "clip_ratio/high_max": 0.0031645570416003466, "clip_ratio/high_mean": 0.0031645570416003466, "clip_ratio/low_mean": 0.004807692370377481, "clip_ratio/low_min": 0.004807692370377481, "clip_ratio/region_mean": 0.007972249411977828, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 78.75, "completions/mean_terminated_length": 78.75, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.040787032805383205, "epoch": 0.0475960959151705, "frac_reward_zero_std": 0.0, "grad_norm": 1.140117883682251, "learning_rate": 6.412121212121213e-06, "loss": -0.0083, "num_tokens": 2669230.0, "reward": 0.9956543445587158, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956543445587158, "reward_meter_std": 0.0006610880373045802, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006611024145968258, "reward_total_composite_mean": 0.9956543445587158, "reward_total_composite_std": 0.0006610880373045802, "reward_total_mean": 0.9956543445587158, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956543445587158, "rewards/meter/std": 0.0006610880373045802, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956543445587158, "rewards/total_composite/std": 0.0006610880373045802, "sampling/importance_sampling_ratio/max": 1.5469458103179932, "sampling/importance_sampling_ratio/mean": 1.0024654865264893, "sampling/importance_sampling_ratio/min": 0.44897207617759705, "sampling/sampling_logp_difference/max": 0.8007946014404297, "sampling/sampling_logp_difference/mean": 0.00999254360795021, "step": 1185 }, { "clip_ratio/high_max": 0.006786415702663362, "clip_ratio/high_mean": 0.006786415702663362, "clip_ratio/low_mean": 0.0010416667209938169, "clip_ratio/low_min": 0.0010416667209938169, "clip_ratio/region_mean": 0.007828082423657179, "completions/clipped_ratio": 0.0, "completions/max_length": 242.0, "completions/max_terminated_length": 242.0, "completions/mean_length": 239.25, "completions/mean_terminated_length": 239.25, "completions/min_length": 236.0, "completions/min_terminated_length": 236.0, "entropy": 0.029086438240483403, "epoch": 0.047636261396955454, "frac_reward_zero_std": 0.0, "grad_norm": 1.235166072845459, "learning_rate": 6.40909090909091e-06, "loss": 0.0014, "num_tokens": 2672128.0, "reward": 0.4595944285392761, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9895985722541809, "reward_meter_std": 0.0064008040353655815, "reward_repeat_penalty_mean": 0.5795454978942871, "reward_repeat_penalty_std": 0.19399181008338928, "reward_std": 0.15526776015758514, "reward_total_composite_mean": 0.4595944285392761, "reward_total_composite_std": 0.15526776015758514, "reward_total_mean": 0.4595944285392761, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9895985722541809, "rewards/meter/std": 0.0064008040353655815, "rewards/repeat_penalty/mean": 0.5795454978942871, "rewards/repeat_penalty/std": 0.19399181008338928, "rewards/total_composite/mean": 0.4595944285392761, "rewards/total_composite/std": 0.15526776015758514, "sampling/importance_sampling_ratio/max": 1.5414388179779053, "sampling/importance_sampling_ratio/mean": 1.0002281665802002, "sampling/importance_sampling_ratio/min": 0.3509824275970459, "sampling/sampling_logp_difference/max": 1.047019124031067, "sampling/sampling_logp_difference/mean": 0.007288929540663958, "step": 1186 }, { "clip_ratio/high_max": 0.0030575180426239967, "clip_ratio/high_mean": 0.0030575180426239967, "clip_ratio/low_mean": 0.006156876712338999, "clip_ratio/low_min": 0.006156876712338999, "clip_ratio/region_mean": 0.009214394754962996, "completions/clipped_ratio": 0.0, "completions/max_length": 377.0, "completions/max_terminated_length": 377.0, "completions/mean_length": 363.25, "completions/mean_terminated_length": 363.25, "completions/min_length": 348.0, "completions/min_terminated_length": 348.0, "entropy": 0.06546153873205185, "epoch": 0.04767642687874041, "frac_reward_zero_std": 0.0, "grad_norm": 1.61380934715271, "learning_rate": 6.406060606060607e-06, "loss": 0.0002, "num_tokens": 2676850.0, "reward": 0.6568292379379272, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9924304485321045, "reward_meter_std": 0.012585594318807125, "reward_repeat_penalty_mean": 0.6617647409439087, "reward_repeat_penalty_std": 0.027230001986026764, "reward_std": 0.030212653800845146, "reward_total_composite_mean": 0.6568292379379272, "reward_total_composite_std": 0.0302126407623291, "reward_total_mean": 0.6568292379379272, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9924304485321045, "rewards/meter/std": 0.012585594318807125, "rewards/repeat_penalty/mean": 0.6617647409439087, "rewards/repeat_penalty/std": 0.027230001986026764, "rewards/total_composite/mean": 0.6568292379379272, "rewards/total_composite/std": 0.0302126407623291, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015922784805298, "sampling/importance_sampling_ratio/min": 0.2921198010444641, "sampling/sampling_logp_difference/max": 1.4292956590652466, "sampling/sampling_logp_difference/mean": 0.012508058920502663, "step": 1187 }, { "clip_ratio/high_max": 0.0013888889516238123, "clip_ratio/high_mean": 0.0013888889516238123, "clip_ratio/low_mean": 0.00206068845000118, "clip_ratio/low_min": 0.00206068845000118, "clip_ratio/region_mean": 0.0034495774016249925, "completions/clipped_ratio": 0.0, "completions/max_length": 368.0, "completions/max_terminated_length": 368.0, "completions/mean_length": 364.0, "completions/mean_terminated_length": 364.0, "completions/min_length": 360.0, "completions/min_terminated_length": 360.0, "entropy": 0.01585481537040323, "epoch": 0.04771659236052536, "frac_reward_zero_std": 0.0, "grad_norm": 1.0509389638900757, "learning_rate": 6.403030303030303e-06, "loss": 0.0066, "num_tokens": 2681346.0, "reward": 0.5992562770843506, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9938761591911316, "reward_meter_std": 0.0003768211172427982, "reward_repeat_penalty_mean": 0.6029411554336548, "reward_repeat_penalty_std": 0.027230001986026764, "reward_std": 0.0272594653069973, "reward_total_composite_mean": 0.5992562770843506, "reward_total_composite_std": 0.027259474620223045, "reward_total_mean": 0.5992562770843506, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9938761591911316, "rewards/meter/std": 0.0003768211172427982, "rewards/repeat_penalty/mean": 0.6029411554336548, "rewards/repeat_penalty/std": 0.027230001986026764, "rewards/total_composite/mean": 0.5992562770843506, "rewards/total_composite/std": 0.027259474620223045, "sampling/importance_sampling_ratio/max": 1.6698790788650513, "sampling/importance_sampling_ratio/mean": 1.0006787776947021, "sampling/importance_sampling_ratio/min": 0.3911004960536957, "sampling/sampling_logp_difference/max": 0.9387906789779663, "sampling/sampling_logp_difference/mean": 0.003385041607543826, "step": 1188 }, { "clip_ratio/high_max": 0.052866541780531406, "clip_ratio/high_mean": 0.052866541780531406, "clip_ratio/low_mean": 0.012431694194674492, "clip_ratio/low_min": 0.012431694194674492, "clip_ratio/region_mean": 0.0652982359752059, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 58.125, "completions/mean_terminated_length": 58.125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.15911316499114037, "epoch": 0.047756757842310316, "frac_reward_zero_std": 0.0, "grad_norm": 5.781359672546387, "learning_rate": 6.4000000000000006e-06, "loss": 0.0367, "num_tokens": 2683179.0, "reward": 0.9804081916809082, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9804081916809082, "reward_meter_std": 0.011680345050990582, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011680349707603455, "reward_total_composite_mean": 0.9804081916809082, "reward_total_composite_std": 0.011680345050990582, "reward_total_mean": 0.9804081916809082, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9804081916809082, "rewards/meter/std": 0.011680345050990582, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9804081916809082, "rewards/total_composite/std": 0.011680345050990582, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9941091537475586, "sampling/importance_sampling_ratio/min": 0.0722494050860405, "sampling/sampling_logp_difference/max": 2.627631187438965, "sampling/sampling_logp_difference/mean": 0.05228231102228165, "step": 1189 }, { "clip_ratio/high_max": 0.008562042959965765, "clip_ratio/high_mean": 0.008562042959965765, "clip_ratio/low_mean": 0.010279399110004306, "clip_ratio/low_min": 0.010279399110004306, "clip_ratio/region_mean": 0.01884144206997007, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 113.625, "completions/mean_terminated_length": 113.625, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.09165592398494482, "epoch": 0.04779692332409527, "frac_reward_zero_std": 0.0, "grad_norm": 6.697084426879883, "learning_rate": 6.396969696969697e-06, "loss": -0.0135, "num_tokens": 2685520.0, "reward": 0.7905905246734619, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9840307831764221, "reward_meter_std": 0.004937502555549145, "reward_repeat_penalty_mean": 0.8035714626312256, "reward_repeat_penalty_std": 0.10628911107778549, "reward_std": 0.10349280387163162, "reward_total_composite_mean": 0.7905905246734619, "reward_total_composite_std": 0.10349281877279282, "reward_total_mean": 0.7905905246734619, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9840307831764221, "rewards/meter/std": 0.004937502555549145, "rewards/repeat_penalty/mean": 0.8035714626312256, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.7905905246734619, "rewards/total_composite/std": 0.10349281877279282, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004918098449707, "sampling/importance_sampling_ratio/min": 0.3153161406517029, "sampling/sampling_logp_difference/max": 1.2992486953735352, "sampling/sampling_logp_difference/mean": 0.023248611018061638, "step": 1190 }, { "clip_ratio/high_max": 0.01574227074161172, "clip_ratio/high_mean": 0.01574227074161172, "clip_ratio/low_mean": 0.013930454850196838, "clip_ratio/low_min": 0.013930454850196838, "clip_ratio/region_mean": 0.029672725591808558, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 117.75, "completions/mean_terminated_length": 117.75, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.16953753679990768, "epoch": 0.047837088805880223, "frac_reward_zero_std": 0.0, "grad_norm": 3.869032859802246, "learning_rate": 6.393939393939394e-06, "loss": -0.0018, "num_tokens": 2687846.0, "reward": 0.8966987133026123, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962882399559021, "reward_meter_std": 0.002757209585979581, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.10691321641206741, "reward_total_composite_mean": 0.8966987133026123, "reward_total_composite_std": 0.10691322386264801, "reward_total_mean": 0.8966987133026123, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962882399559021, "rewards/meter/std": 0.002757209585979581, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8966987133026123, "rewards/total_composite/std": 0.10691322386264801, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0035390853881836, "sampling/importance_sampling_ratio/min": 0.24718700349330902, "sampling/sampling_logp_difference/max": 1.3976101875305176, "sampling/sampling_logp_difference/mean": 0.03010808303952217, "step": 1191 }, { "clip_ratio/high_max": 0.007911392254754901, "clip_ratio/high_mean": 0.007911392254754901, "clip_ratio/low_mean": 0.0015822785208001733, "clip_ratio/low_min": 0.0015822785208001733, "clip_ratio/region_mean": 0.009493670775555074, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 79.0, "completions/mean_terminated_length": 79.0, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.033365981886163354, "epoch": 0.04787725428766518, "frac_reward_zero_std": 0.0, "grad_norm": 5.973384380340576, "learning_rate": 6.390909090909091e-06, "loss": -0.0028, "num_tokens": 2689734.0, "reward": 0.994601309299469, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994601309299469, "reward_meter_std": 0.003018211107701063, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0030182143673300743, "reward_total_composite_mean": 0.994601309299469, "reward_total_composite_std": 0.003018211107701063, "reward_total_mean": 0.994601309299469, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994601309299469, "rewards/meter/std": 0.003018211107701063, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994601309299469, "rewards/total_composite/std": 0.003018211107701063, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0011334419250488, "sampling/importance_sampling_ratio/min": 0.3428289294242859, "sampling/sampling_logp_difference/max": 1.7237977981567383, "sampling/sampling_logp_difference/mean": 0.01802201196551323, "step": 1192 }, { "clip_ratio/high_max": 0.027820809744298458, "clip_ratio/high_mean": 0.027820809744298458, "clip_ratio/low_mean": 0.015306122601032257, "clip_ratio/low_min": 0.015306122601032257, "clip_ratio/region_mean": 0.043126932345330715, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 61.125, "completions/mean_terminated_length": 61.125, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.16935504972934723, "epoch": 0.04791741976945013, "frac_reward_zero_std": 0.0, "grad_norm": 7.629351615905762, "learning_rate": 6.387878787878789e-06, "loss": -0.0592, "num_tokens": 2691559.0, "reward": 0.8711988925933838, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8711988925933838, "reward_meter_std": 0.3447757959365845, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3447757661342621, "reward_total_composite_mean": 0.8711988925933838, "reward_total_composite_std": 0.3447757959365845, "reward_total_mean": 0.8711988925933838, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8711988925933838, "rewards/meter/std": 0.3447757959365845, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8711988925933838, "rewards/total_composite/std": 0.3447757959365845, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0055021047592163, "sampling/importance_sampling_ratio/min": 0.3974994122982025, "sampling/sampling_logp_difference/max": 0.9225618839263916, "sampling/sampling_logp_difference/mean": 0.03702111169695854, "step": 1193 }, { "clip_ratio/high_max": 0.02205510064959526, "clip_ratio/high_mean": 0.02205510064959526, "clip_ratio/low_mean": 0.0029069767333567142, "clip_ratio/low_min": 0.0029069767333567142, "clip_ratio/region_mean": 0.024962077382951975, "completions/clipped_ratio": 0.0, "completions/max_length": 89.0, "completions/max_terminated_length": 89.0, "completions/mean_length": 83.625, "completions/mean_terminated_length": 83.625, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.12350557837635279, "epoch": 0.047957585251235085, "frac_reward_zero_std": 0.0, "grad_norm": 4.607259750366211, "learning_rate": 6.384848484848485e-06, "loss": 0.0146, "num_tokens": 2693468.0, "reward": 0.9963547587394714, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963547587394714, "reward_meter_std": 0.0017178525449708104, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0017178438138216734, "reward_total_composite_mean": 0.9963547587394714, "reward_total_composite_std": 0.0017178525449708104, "reward_total_mean": 0.9963547587394714, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963547587394714, "rewards/meter/std": 0.0017178525449708104, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963547587394714, "rewards/total_composite/std": 0.0017178525449708104, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0033197402954102, "sampling/importance_sampling_ratio/min": 0.16897356510162354, "sampling/sampling_logp_difference/max": 1.778012990951538, "sampling/sampling_logp_difference/mean": 0.0324261412024498, "step": 1194 }, { "clip_ratio/high_max": 0.011043233331292868, "clip_ratio/high_mean": 0.011043233331292868, "clip_ratio/low_mean": 0.006657268386334181, "clip_ratio/low_min": 0.006657268386334181, "clip_ratio/region_mean": 0.01770050171762705, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 56.0, "completions/mean_terminated_length": 56.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.11227265931665897, "epoch": 0.04799775073302004, "frac_reward_zero_std": 0.0, "grad_norm": 6.79469108581543, "learning_rate": 6.381818181818182e-06, "loss": -0.0016, "num_tokens": 2695188.0, "reward": 0.07255840301513672, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.08778245747089386, "reward_meter_std": 0.11504703015089035, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.17817415297031403, "reward_std": 0.0835263803601265, "reward_total_composite_mean": 0.07255840301513672, "reward_total_composite_std": 0.08352638781070709, "reward_total_mean": 0.07255840301513672, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.08778245747089386, "rewards/meter/std": 0.11504703015089035, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.17817415297031403, "rewards/total_composite/mean": 0.07255840301513672, "rewards/total_composite/std": 0.08352638781070709, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0030452013015747, "sampling/importance_sampling_ratio/min": 0.37429752945899963, "sampling/sampling_logp_difference/max": 0.9827042818069458, "sampling/sampling_logp_difference/mean": 0.022929562255740166, "step": 1195 }, { "clip_ratio/high_max": 0.012287983670830727, "clip_ratio/high_mean": 0.012287983670830727, "clip_ratio/low_mean": 0.023972635506652296, "clip_ratio/low_min": 0.023972635506652296, "clip_ratio/region_mean": 0.03626061917748302, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 92.875, "completions/mean_terminated_length": 92.875, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.14395452290773392, "epoch": 0.04803791621480499, "frac_reward_zero_std": 0.0, "grad_norm": 5.267541408538818, "learning_rate": 6.37878787878788e-06, "loss": 0.0143, "num_tokens": 2697251.0, "reward": 0.8394116759300232, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9628203511238098, "reward_meter_std": 0.07420913130044937, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.0950227826833725, "reward_total_composite_mean": 0.8394116759300232, "reward_total_composite_std": 0.0950227826833725, "reward_total_mean": 0.8394116759300232, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9628203511238098, "rewards/meter/std": 0.07420913130044937, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8394116759300232, "rewards/total_composite/std": 0.0950227826833725, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9947278499603271, "sampling/importance_sampling_ratio/min": 0.20306822657585144, "sampling/sampling_logp_difference/max": 1.5942132472991943, "sampling/sampling_logp_difference/mean": 0.046500932425260544, "step": 1196 }, { "clip_ratio/high_max": 0.004789536877069622, "clip_ratio/high_mean": 0.004789536877069622, "clip_ratio/low_mean": 0.0029802137287333608, "clip_ratio/low_min": 0.0029802137287333608, "clip_ratio/region_mean": 0.007769750605802983, "completions/clipped_ratio": 0.0, "completions/max_length": 474.0, "completions/max_terminated_length": 474.0, "completions/mean_length": 419.375, "completions/mean_terminated_length": 419.375, "completions/min_length": 402.0, "completions/min_terminated_length": 402.0, "entropy": 0.04937975201755762, "epoch": 0.048078081696589954, "frac_reward_zero_std": 0.0, "grad_norm": 0.9943876266479492, "learning_rate": 6.375757575757576e-06, "loss": 0.0431, "num_tokens": 2702278.0, "reward": 0.3697792887687683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6333333253860474, "reward_count_adherence_std": 0.0942809134721756, "reward_meter_mean": 0.8696904182434082, "reward_meter_std": 0.3263057768344879, "reward_repeat_penalty_mean": 0.6529605388641357, "reward_repeat_penalty_std": 0.05881762132048607, "reward_std": 0.14553934335708618, "reward_total_composite_mean": 0.3697792887687683, "reward_total_composite_std": 0.14553934335708618, "reward_total_mean": 0.3697792887687683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6333333253860474, "rewards/count_adherence/std": 0.0942809134721756, "rewards/meter/mean": 0.8696904182434082, "rewards/meter/std": 0.3263057768344879, "rewards/repeat_penalty/mean": 0.6529605388641357, "rewards/repeat_penalty/std": 0.05881762132048607, "rewards/total_composite/mean": 0.3697792887687683, "rewards/total_composite/std": 0.14553934335708618, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0019516944885254, "sampling/importance_sampling_ratio/min": 0.3320879638195038, "sampling/sampling_logp_difference/max": 1.1023553609848022, "sampling/sampling_logp_difference/mean": 0.009604532271623611, "step": 1197 }, { "clip_ratio/high_max": 0.006082487525418401, "clip_ratio/high_mean": 0.006082487525418401, "clip_ratio/low_mean": 0.014116051141172647, "clip_ratio/low_min": 0.014116051141172647, "clip_ratio/region_mean": 0.020198538666591048, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.5, "completions/mean_terminated_length": 61.5, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.31682545877993107, "epoch": 0.04811824717837491, "frac_reward_zero_std": 0.0, "grad_norm": 6.456137657165527, "learning_rate": 6.372727272727274e-06, "loss": 0.0032, "num_tokens": 2703954.0, "reward": 0.3377665877342224, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3377665877342224, "reward_meter_std": 0.33047032356262207, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.33047032356262207, "reward_total_composite_mean": 0.3377665877342224, "reward_total_composite_std": 0.33047032356262207, "reward_total_mean": 0.3377665877342224, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3377665877342224, "rewards/meter/std": 0.33047032356262207, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3377665877342224, "rewards/total_composite/std": 0.33047032356262207, "sampling/importance_sampling_ratio/max": 1.5698946714401245, "sampling/importance_sampling_ratio/mean": 1.0035641193389893, "sampling/importance_sampling_ratio/min": 0.19029447436332703, "sampling/sampling_logp_difference/max": 1.6591825485229492, "sampling/sampling_logp_difference/mean": 0.0402526780962944, "step": 1198 }, { "clip_ratio/high_max": 0.0031447785440832376, "clip_ratio/high_mean": 0.0031447785440832376, "clip_ratio/low_mean": 0.004537329194135964, "clip_ratio/low_min": 0.004537329194135964, "clip_ratio/region_mean": 0.0076821077382192016, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 81.125, "completions/mean_terminated_length": 81.125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.04470323724672198, "epoch": 0.04815841266015986, "frac_reward_zero_std": 0.0, "grad_norm": 4.539976596832275, "learning_rate": 6.3696969696969706e-06, "loss": 0.0199, "num_tokens": 2705923.0, "reward": 0.5489683747291565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5489683747291565, "reward_meter_std": 0.4779263734817505, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4779263436794281, "reward_total_composite_mean": 0.5489683747291565, "reward_total_composite_std": 0.4779263734817505, "reward_total_mean": 0.5489683747291565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5489683747291565, "rewards/meter/std": 0.4779263734817505, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5489683747291565, "rewards/total_composite/std": 0.4779263734817505, "sampling/importance_sampling_ratio/max": 1.4673889875411987, "sampling/importance_sampling_ratio/mean": 1.0014384984970093, "sampling/importance_sampling_ratio/min": 0.37336301803588867, "sampling/sampling_logp_difference/max": 0.9852040410041809, "sampling/sampling_logp_difference/mean": 0.010573667474091053, "step": 1199 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.0147115015424788, "clip_ratio/low_min": 0.0147115015424788, "clip_ratio/region_mean": 0.01676068175584078, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 59.875, "completions/mean_terminated_length": 59.875, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.09853810910135508, "epoch": 0.048198578141944816, "frac_reward_zero_std": 0.0, "grad_norm": 13.826656341552734, "learning_rate": 6.366666666666668e-06, "loss": 0.018, "num_tokens": 2707810.0, "reward": 0.8469290137290955, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8469290137290955, "reward_meter_std": 0.20266500115394592, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.20266498625278473, "reward_total_composite_mean": 0.8469290137290955, "reward_total_composite_std": 0.20266500115394592, "reward_total_mean": 0.8469290137290955, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8469290137290955, "rewards/meter/std": 0.20266500115394592, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8469290137290955, "rewards/total_composite/std": 0.20266500115394592, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0088112354278564, "sampling/importance_sampling_ratio/min": 0.15658754110336304, "sampling/sampling_logp_difference/max": 1.854140043258667, "sampling/sampling_logp_difference/mean": 0.036583464592695236, "step": 1200 }, { "epoch": 0.048198578141944816, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 365.9230769230769, "eval_completions/max_terminated_length": 365.9230769230769, "eval_completions/mean_length": 210.4903846153846, "eval_completions/mean_terminated_length": 210.4903846153846, "eval_completions/min_length": 63.69230769230769, "eval_completions/min_terminated_length": 63.69230769230769, "eval_entropy": 0.07712389929936482, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 2707810.0, "eval_reward": 0.4327001617505, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.8891764970926138, "eval_reward_count_adherence_std": 0.11591204485067955, "eval_reward_meter_mean": 0.675024688243866, "eval_reward_meter_std": 0.4085804086465102, "eval_reward_repeat_penalty_mean": 0.7145149661944463, "eval_reward_repeat_penalty_std": 0.17899919931705183, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.4327001617505, "eval_reward_total_composite_std": 0.3292730886202592, "eval_reward_total_mean": 0.4327001617505, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.8891764970926138, "eval_rewards/count_adherence/std": 0.11591204485067955, "eval_rewards/meter/mean": 0.675024688243866, "eval_rewards/meter/std": 0.4085804086465102, "eval_rewards/repeat_penalty/mean": 0.7145149661944463, "eval_rewards/repeat_penalty/std": 0.17899919931705183, "eval_rewards/total_composite/mean": 0.4327001617505, "eval_rewards/total_composite/std": 0.3292730886202592, "eval_runtime": 69.5055, "eval_samples_per_second": 1.496, "eval_sampling/importance_sampling_ratio/max": 1.3788938980836134, "eval_sampling/importance_sampling_ratio/mean": 1.0017014145851135, "eval_sampling/importance_sampling_ratio/min": 0.4270020562868852, "eval_sampling/sampling_logp_difference/max": 0.8752486155583308, "eval_sampling/sampling_logp_difference/mean": 0.009530584721897658, "eval_steps_per_second": 0.187, "step": 1200 }, { "clip_ratio/high_max": 0.0015822785208001733, "clip_ratio/high_mean": 0.0015822785208001733, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0015822785208001733, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 79.125, "completions/mean_terminated_length": 79.125, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.022257187170907855, "epoch": 0.04823874362372977, "frac_reward_zero_std": 0.0, "grad_norm": 1.0187959671020508, "learning_rate": 6.363636363636364e-06, "loss": 0.0014, "num_tokens": 2709651.0, "reward": 0.9956833124160767, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956833124160767, "reward_meter_std": 4.2210078390780836e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.2219100578222424e-05, "reward_total_composite_mean": 0.9956833124160767, "reward_total_composite_std": 4.2210078390780836e-05, "reward_total_mean": 0.9956833124160767, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956833124160767, "rewards/meter/std": 4.2210078390780836e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956833124160767, "rewards/total_composite/std": 4.2210078390780836e-05, "sampling/importance_sampling_ratio/max": 1.1379395723342896, "sampling/importance_sampling_ratio/mean": 1.0009052753448486, "sampling/importance_sampling_ratio/min": 0.5761146545410156, "sampling/sampling_logp_difference/max": 0.5514485836029053, "sampling/sampling_logp_difference/mean": 0.0033457186073064804, "step": 1201 }, { "clip_ratio/high_max": 0.012096773833036423, "clip_ratio/high_mean": 0.012096773833036423, "clip_ratio/low_mean": 0.03160349419340491, "clip_ratio/low_min": 0.03160349419340491, "clip_ratio/region_mean": 0.043700268026441336, "completions/clipped_ratio": 0.0, "completions/max_length": 31.0, "completions/max_terminated_length": 31.0, "completions/mean_length": 29.875, "completions/mean_terminated_length": 29.875, "completions/min_length": 27.0, "completions/min_terminated_length": 27.0, "entropy": 0.21921741589903831, "epoch": 0.048278909105514724, "frac_reward_zero_std": 0.0, "grad_norm": 8.3601655960083, "learning_rate": 6.3606060606060615e-06, "loss": -0.0406, "num_tokens": 2711138.0, "reward": 0.644133985042572, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.644133985042572, "reward_meter_std": 0.4584461748600006, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4584461748600006, "reward_total_composite_mean": 0.644133985042572, "reward_total_composite_std": 0.4584461748600006, "reward_total_mean": 0.644133985042572, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.644133985042572, "rewards/meter/std": 0.4584461748600006, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.644133985042572, "rewards/total_composite/std": 0.4584461748600006, "sampling/importance_sampling_ratio/max": 1.6411713361740112, "sampling/importance_sampling_ratio/mean": 1.0082812309265137, "sampling/importance_sampling_ratio/min": 0.3355531394481659, "sampling/sampling_logp_difference/max": 1.0919749736785889, "sampling/sampling_logp_difference/mean": 0.034943368285894394, "step": 1202 }, { "clip_ratio/high_max": 0.008405112195760012, "clip_ratio/high_mean": 0.008405112195760012, "clip_ratio/low_mean": 0.04342324007302523, "clip_ratio/low_min": 0.04342324007302523, "clip_ratio/region_mean": 0.05182835226878524, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 60.5, "completions/mean_terminated_length": 60.5, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.37029214575886726, "epoch": 0.04831907458729968, "frac_reward_zero_std": 0.0, "grad_norm": 4.518101215362549, "learning_rate": 6.357575757575758e-06, "loss": 0.0002, "num_tokens": 2712910.0, "reward": 0.203798308968544, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.203798308968544, "reward_meter_std": 0.2958202362060547, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2958202362060547, "reward_total_composite_mean": 0.203798308968544, "reward_total_composite_std": 0.2958202362060547, "reward_total_mean": 0.203798308968544, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.203798308968544, "rewards/meter/std": 0.2958202362060547, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.203798308968544, "rewards/total_composite/std": 0.2958202362060547, "sampling/importance_sampling_ratio/max": 1.886362910270691, "sampling/importance_sampling_ratio/mean": 1.004960060119629, "sampling/importance_sampling_ratio/min": 0.18201525509357452, "sampling/sampling_logp_difference/max": 1.703664779663086, "sampling/sampling_logp_difference/mean": 0.05345018953084946, "step": 1203 }, { "clip_ratio/high_max": 0.003989361692219973, "clip_ratio/high_mean": 0.003989361692219973, "clip_ratio/low_mean": 0.01712075702380389, "clip_ratio/low_min": 0.01712075702380389, "clip_ratio/region_mean": 0.021110118716023862, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 94.125, "completions/mean_terminated_length": 94.125, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.06619423814117908, "epoch": 0.04835924006908463, "frac_reward_zero_std": 0.0, "grad_norm": 3.396805763244629, "learning_rate": 6.354545454545455e-06, "loss": 0.004, "num_tokens": 2715103.0, "reward": 0.8099536895751953, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9817042350769043, "reward_meter_std": 0.002063595922663808, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.070185586810112, "reward_total_composite_mean": 0.8099536895751953, "reward_total_composite_std": 0.070185586810112, "reward_total_mean": 0.8099536895751953, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9817042350769043, "rewards/meter/std": 0.002063595922663808, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8099536895751953, "rewards/total_composite/std": 0.070185586810112, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0024614334106445, "sampling/importance_sampling_ratio/min": 0.29409554600715637, "sampling/sampling_logp_difference/max": 1.2238504886627197, "sampling/sampling_logp_difference/mean": 0.022159146144986153, "step": 1204 }, { "clip_ratio/high_max": 0.014675587648525834, "clip_ratio/high_mean": 0.014675587648525834, "clip_ratio/low_mean": 0.004518072120845318, "clip_ratio/low_min": 0.004518072120845318, "clip_ratio/region_mean": 0.019193659769371152, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 78.5, "completions/mean_terminated_length": 78.5, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.18105985596776009, "epoch": 0.048399405550869586, "frac_reward_zero_std": 0.0, "grad_norm": 4.248809814453125, "learning_rate": 6.3515151515151516e-06, "loss": 0.0175, "num_tokens": 2717019.0, "reward": 0.9937853813171387, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9937853813171387, "reward_meter_std": 0.0054353163577616215, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0054353224113583565, "reward_total_composite_mean": 0.9937853813171387, "reward_total_composite_std": 0.0054353163577616215, "reward_total_mean": 0.9937853813171387, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9937853813171387, "rewards/meter/std": 0.0054353163577616215, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9937853813171387, "rewards/total_composite/std": 0.0054353163577616215, "sampling/importance_sampling_ratio/max": 1.6284278631210327, "sampling/importance_sampling_ratio/mean": 1.0043216943740845, "sampling/importance_sampling_ratio/min": 0.2697693705558777, "sampling/sampling_logp_difference/max": 1.310187816619873, "sampling/sampling_logp_difference/mean": 0.03144395351409912, "step": 1205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/region_mean": 0.001623376621864736, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 77.75, "completions/mean_terminated_length": 77.75, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.022122491151094437, "epoch": 0.04843957103265454, "frac_reward_zero_std": 0.0, "grad_norm": 2.2625913619995117, "learning_rate": 6.34848484848485e-06, "loss": -0.0049, "num_tokens": 2718833.0, "reward": 0.9951738119125366, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9951738119125366, "reward_meter_std": 0.00032568458118475974, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00032568458118475974, "reward_total_composite_mean": 0.9951738119125366, "reward_total_composite_std": 0.00032568458118475974, "reward_total_mean": 0.9951738119125366, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9951738119125366, "rewards/meter/std": 0.00032568458118475974, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951738119125366, "rewards/total_composite/std": 0.00032568458118475974, "sampling/importance_sampling_ratio/max": 1.3814133405685425, "sampling/importance_sampling_ratio/mean": 1.0006308555603027, "sampling/importance_sampling_ratio/min": 0.636854887008667, "sampling/sampling_logp_difference/max": 0.4512134790420532, "sampling/sampling_logp_difference/mean": 0.004397572483867407, "step": 1206 }, { "clip_ratio/high_max": 0.013700352283194661, "clip_ratio/high_mean": 0.013700352283194661, "clip_ratio/low_mean": 0.005122669972479343, "clip_ratio/low_min": 0.005122669972479343, "clip_ratio/region_mean": 0.018823022255674005, "completions/clipped_ratio": 0.0, "completions/max_length": 130.0, "completions/max_terminated_length": 130.0, "completions/mean_length": 124.5, "completions/mean_terminated_length": 124.5, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.1076383925974369, "epoch": 0.04847973651443949, "frac_reward_zero_std": 0.0, "grad_norm": 7.367405891418457, "learning_rate": 6.345454545454546e-06, "loss": -0.005, "num_tokens": 2721333.0, "reward": 0.5791441202163696, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8035296201705933, "reward_meter_std": 0.35723742842674255, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.24694089591503143, "reward_total_composite_mean": 0.5791441202163696, "reward_total_composite_std": 0.24694091081619263, "reward_total_mean": 0.5791441202163696, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8035296201705933, "rewards/meter/std": 0.35723742842674255, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.5791441202163696, "rewards/total_composite/std": 0.24694091081619263, "sampling/importance_sampling_ratio/max": 1.7591142654418945, "sampling/importance_sampling_ratio/mean": 1.0008379220962524, "sampling/importance_sampling_ratio/min": 0.20985685288906097, "sampling/sampling_logp_difference/max": 1.5613296031951904, "sampling/sampling_logp_difference/mean": 0.022883908823132515, "step": 1207 }, { "clip_ratio/high_max": 0.016762672923505306, "clip_ratio/high_mean": 0.016762672923505306, "clip_ratio/low_mean": 0.03671912616118789, "clip_ratio/low_min": 0.03671912616118789, "clip_ratio/region_mean": 0.05348179908469319, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.3939699362963438, "epoch": 0.04851990199622445, "frac_reward_zero_std": 0.0, "grad_norm": 5.575094699859619, "learning_rate": 6.342424242424243e-06, "loss": -0.0147, "num_tokens": 2723045.0, "reward": 0.3075833320617676, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3075833320617676, "reward_meter_std": 0.3925836980342865, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3925836980342865, "reward_total_composite_mean": 0.3075833320617676, "reward_total_composite_std": 0.3925836980342865, "reward_total_mean": 0.3075833320617676, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3075833320617676, "rewards/meter/std": 0.3925836980342865, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3075833320617676, "rewards/total_composite/std": 0.3925836980342865, "sampling/importance_sampling_ratio/max": 1.9152469635009766, "sampling/importance_sampling_ratio/mean": 0.9958197474479675, "sampling/importance_sampling_ratio/min": 0.20231691002845764, "sampling/sampling_logp_difference/max": 1.5979199409484863, "sampling/sampling_logp_difference/mean": 0.07028155028820038, "step": 1208 }, { "clip_ratio/high_max": 0.010197082534432411, "clip_ratio/high_mean": 0.010197082534432411, "clip_ratio/low_mean": 0.004963513347320259, "clip_ratio/low_min": 0.004963513347320259, "clip_ratio/region_mean": 0.01516059588175267, "completions/clipped_ratio": 0.0, "completions/max_length": 246.0, "completions/max_terminated_length": 246.0, "completions/mean_length": 230.25, "completions/mean_terminated_length": 230.25, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.10937195271253586, "epoch": 0.0485600674780094, "frac_reward_zero_std": 0.0, "grad_norm": 2.477656364440918, "learning_rate": 6.33939393939394e-06, "loss": -0.0039, "num_tokens": 2726311.0, "reward": 0.5340035557746887, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955352544784546, "reward_meter_std": 0.0017380027566105127, "reward_repeat_penalty_mean": 0.6704545617103577, "reward_repeat_penalty_std": 0.13690368831157684, "reward_std": 0.10914173722267151, "reward_total_composite_mean": 0.5340035557746887, "reward_total_composite_std": 0.10914173722267151, "reward_total_mean": 0.5340035557746887, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955352544784546, "rewards/meter/std": 0.0017380027566105127, "rewards/repeat_penalty/mean": 0.6704545617103577, "rewards/repeat_penalty/std": 0.13690368831157684, "rewards/total_composite/mean": 0.5340035557746887, "rewards/total_composite/std": 0.10914173722267151, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0021883249282837, "sampling/importance_sampling_ratio/min": 0.15673324465751648, "sampling/sampling_logp_difference/max": 1.8532099723815918, "sampling/sampling_logp_difference/mean": 0.020786819979548454, "step": 1209 }, { "clip_ratio/high_max": 0.001087038719560951, "clip_ratio/high_mean": 0.001087038719560951, "clip_ratio/low_mean": 0.0016003104392439127, "clip_ratio/low_min": 0.0016003104392439127, "clip_ratio/region_mean": 0.0026873491588048637, "completions/clipped_ratio": 0.0, "completions/max_length": 235.0, "completions/max_terminated_length": 235.0, "completions/mean_length": 232.125, "completions/mean_terminated_length": 232.125, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.03385857096873224, "epoch": 0.048600232959794355, "frac_reward_zero_std": 0.0, "grad_norm": 0.6899715662002563, "learning_rate": 6.336363636363637e-06, "loss": 0.0075, "num_tokens": 2729552.0, "reward": 0.36228621006011963, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961996078491211, "reward_meter_std": 0.0006288865115493536, "reward_repeat_penalty_mean": 0.4545454680919647, "reward_repeat_penalty_std": 0.11902794241905212, "reward_std": 0.09505681693553925, "reward_total_composite_mean": 0.36228621006011963, "reward_total_composite_std": 0.09505681693553925, "reward_total_mean": 0.36228621006011963, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961996078491211, "rewards/meter/std": 0.0006288865115493536, "rewards/repeat_penalty/mean": 0.4545454680919647, "rewards/repeat_penalty/std": 0.11902794241905212, "rewards/total_composite/mean": 0.36228621006011963, "rewards/total_composite/std": 0.09505681693553925, "sampling/importance_sampling_ratio/max": 1.5110461711883545, "sampling/importance_sampling_ratio/mean": 1.0020835399627686, "sampling/importance_sampling_ratio/min": 0.3738209903240204, "sampling/sampling_logp_difference/max": 0.983978271484375, "sampling/sampling_logp_difference/mean": 0.005651684943586588, "step": 1210 }, { "clip_ratio/high_max": 0.005265707382932305, "clip_ratio/high_mean": 0.005265707382932305, "clip_ratio/low_mean": 0.00556039166986011, "clip_ratio/low_min": 0.00556039166986011, "clip_ratio/region_mean": 0.010826099052792415, "completions/clipped_ratio": 0.0, "completions/max_length": 320.0, "completions/max_terminated_length": 320.0, "completions/mean_length": 305.625, "completions/mean_terminated_length": 305.625, "completions/min_length": 289.0, "completions/min_terminated_length": 289.0, "entropy": 0.04799028439447284, "epoch": 0.04864039844157931, "frac_reward_zero_std": 0.0, "grad_norm": 2.939035177230835, "learning_rate": 6.333333333333333e-06, "loss": 0.0165, "num_tokens": 2733597.0, "reward": 0.6673710942268372, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9444444179534912, "reward_count_adherence_std": 0.059391383081674576, "reward_meter_mean": 0.9956304430961609, "reward_meter_std": 0.0010864537907764316, "reward_repeat_penalty_mean": 0.7098382711410522, "reward_repeat_penalty_std": 0.09350526332855225, "reward_std": 0.0976538211107254, "reward_total_composite_mean": 0.6673710942268372, "reward_total_composite_std": 0.0976538211107254, "reward_total_mean": 0.6673710942268372, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9444444179534912, "rewards/count_adherence/std": 0.059391383081674576, "rewards/meter/mean": 0.9956304430961609, "rewards/meter/std": 0.0010864537907764316, "rewards/repeat_penalty/mean": 0.7098382711410522, "rewards/repeat_penalty/std": 0.09350526332855225, "rewards/total_composite/mean": 0.6673710942268372, "rewards/total_composite/std": 0.0976538211107254, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001654028892517, "sampling/importance_sampling_ratio/min": 0.06939523667097092, "sampling/sampling_logp_difference/max": 2.966305732727051, "sampling/sampling_logp_difference/mean": 0.01374336052685976, "step": 1211 }, { "clip_ratio/high_max": 0.004032257944345474, "clip_ratio/high_mean": 0.004032257944345474, "clip_ratio/low_mean": 0.010017690248787403, "clip_ratio/low_min": 0.010017690248787403, "clip_ratio/region_mean": 0.014049948193132877, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 62.125, "completions/mean_terminated_length": 62.125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.05741153517737985, "epoch": 0.04868056392336426, "frac_reward_zero_std": 0.0, "grad_norm": 1.3120684623718262, "learning_rate": 6.330303030303031e-06, "loss": 0.0002, "num_tokens": 2735462.0, "reward": 0.993018627166748, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.993018627166748, "reward_meter_std": 0.0011156242107972503, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0011156280525028706, "reward_total_composite_mean": 0.993018627166748, "reward_total_composite_std": 0.0011156242107972503, "reward_total_mean": 0.993018627166748, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.993018627166748, "rewards/meter/std": 0.0011156242107972503, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993018627166748, "rewards/total_composite/std": 0.0011156242107972503, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0012403726577759, "sampling/importance_sampling_ratio/min": 0.4961382746696472, "sampling/sampling_logp_difference/max": 1.2096130847930908, "sampling/sampling_logp_difference/mean": 0.016475966200232506, "step": 1212 }, { "clip_ratio/high_max": 0.007692307815887034, "clip_ratio/high_mean": 0.007692307815887034, "clip_ratio/low_mean": 0.012202381622046232, "clip_ratio/low_min": 0.012202381622046232, "clip_ratio/region_mean": 0.019894689437933266, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 61.875, "completions/mean_terminated_length": 61.875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.08121698070317507, "epoch": 0.04872072940514922, "frac_reward_zero_std": 0.0, "grad_norm": 5.671288967132568, "learning_rate": 6.327272727272727e-06, "loss": -0.0134, "num_tokens": 2737205.0, "reward": 0.9939371347427368, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939371347427368, "reward_meter_std": 0.0013982808450236917, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0013982723467051983, "reward_total_composite_mean": 0.9939371347427368, "reward_total_composite_std": 0.0013982808450236917, "reward_total_mean": 0.9939371347427368, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939371347427368, "rewards/meter/std": 0.0013982808450236917, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9939371347427368, "rewards/total_composite/std": 0.0013982808450236917, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9959902763366699, "sampling/importance_sampling_ratio/min": 0.3034030497074127, "sampling/sampling_logp_difference/max": 1.1926932334899902, "sampling/sampling_logp_difference/mean": 0.02799602597951889, "step": 1213 }, { "clip_ratio/high_max": 0.002272886282298714, "clip_ratio/high_mean": 0.002272886282298714, "clip_ratio/low_mean": 0.0023547938617412, "clip_ratio/low_min": 0.0023547938617412, "clip_ratio/region_mean": 0.004627680144039914, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 323.875, "completions/mean_terminated_length": 323.875, "completions/min_length": 316.0, "completions/min_terminated_length": 316.0, "entropy": 0.0330718404147774, "epoch": 0.04876089488693417, "frac_reward_zero_std": 0.0, "grad_norm": 1.0214275121688843, "learning_rate": 6.324242424242425e-06, "loss": -0.0101, "num_tokens": 2741692.0, "reward": 0.5024136304855347, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.035355325788259506, "reward_meter_mean": 0.9967888593673706, "reward_meter_std": 0.0008384960819967091, "reward_repeat_penalty_mean": 0.6199448704719543, "reward_repeat_penalty_std": 0.08540944010019302, "reward_std": 0.07446449995040894, "reward_total_composite_mean": 0.5024136304855347, "reward_total_composite_std": 0.07446450740098953, "reward_total_mean": 0.5024136304855347, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.035355325788259506, "rewards/meter/mean": 0.9967888593673706, "rewards/meter/std": 0.0008384960819967091, "rewards/repeat_penalty/mean": 0.6199448704719543, "rewards/repeat_penalty/std": 0.08540944010019302, "rewards/total_composite/mean": 0.5024136304855347, "rewards/total_composite/std": 0.07446450740098953, "sampling/importance_sampling_ratio/max": 1.6564884185791016, "sampling/importance_sampling_ratio/mean": 1.0000638961791992, "sampling/importance_sampling_ratio/min": 0.10816873610019684, "sampling/sampling_logp_difference/max": 2.224062919616699, "sampling/sampling_logp_difference/mean": 0.007462342269718647, "step": 1214 }, { "clip_ratio/high_max": 0.006238285917788744, "clip_ratio/high_mean": 0.006238285917788744, "clip_ratio/low_mean": 0.0077110049314796925, "clip_ratio/low_min": 0.0077110049314796925, "clip_ratio/region_mean": 0.013949290849268436, "completions/clipped_ratio": 0.0, "completions/max_length": 246.0, "completions/max_terminated_length": 246.0, "completions/mean_length": 241.625, "completions/mean_terminated_length": 241.625, "completions/min_length": 235.0, "completions/min_terminated_length": 235.0, "entropy": 0.061672994401305914, "epoch": 0.048801060368719125, "frac_reward_zero_std": 0.0, "grad_norm": 2.1483280658721924, "learning_rate": 6.3212121212121216e-06, "loss": 0.0064, "num_tokens": 2745177.0, "reward": 0.7299137115478516, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970427751541138, "reward_meter_std": 0.0011380411451682448, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.0739355981349945, "reward_std": 0.07295886427164078, "reward_total_composite_mean": 0.7299137115478516, "reward_total_composite_std": 0.07295885682106018, "reward_total_mean": 0.7299137115478516, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970427751541138, "rewards/meter/std": 0.0011380411451682448, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.0739355981349945, "rewards/total_composite/mean": 0.7299137115478516, "rewards/total_composite/std": 0.07295885682106018, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000618815422058, "sampling/importance_sampling_ratio/min": 0.2422354817390442, "sampling/sampling_logp_difference/max": 1.4178450107574463, "sampling/sampling_logp_difference/mean": 0.016915393993258476, "step": 1215 }, { "clip_ratio/high_max": 0.006018434185534716, "clip_ratio/high_mean": 0.006018434185534716, "clip_ratio/low_mean": 0.003024590201675892, "clip_ratio/low_min": 0.003024590201675892, "clip_ratio/region_mean": 0.009043024387210608, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 124.125, "completions/mean_terminated_length": 124.125, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.03512992733158171, "epoch": 0.04884122585050408, "frac_reward_zero_std": 0.0, "grad_norm": 1.560193657875061, "learning_rate": 6.318181818181819e-06, "loss": -0.0023, "num_tokens": 2747538.0, "reward": 0.7628991007804871, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99362713098526, "reward_meter_std": 0.0015980940079316497, "reward_repeat_penalty_mean": 0.7678571343421936, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07266796380281448, "reward_total_composite_mean": 0.7628991007804871, "reward_total_composite_std": 0.07266794145107269, "reward_total_mean": 0.7628991007804871, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99362713098526, "rewards/meter/std": 0.0015980940079316497, "rewards/repeat_penalty/mean": 0.7678571343421936, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.7628991007804871, "rewards/total_composite/std": 0.07266794145107269, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9995880722999573, "sampling/importance_sampling_ratio/min": 0.3855396807193756, "sampling/sampling_logp_difference/max": 0.9531111717224121, "sampling/sampling_logp_difference/mean": 0.010110512375831604, "step": 1216 }, { "clip_ratio/high_max": 0.00835704104974866, "clip_ratio/high_mean": 0.00835704104974866, "clip_ratio/low_mean": 0.009840250364504755, "clip_ratio/low_min": 0.009840250364504755, "clip_ratio/region_mean": 0.018197291414253414, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 75.75, "completions/mean_terminated_length": 75.75, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.20184578280895948, "epoch": 0.04888139133228903, "frac_reward_zero_std": 0.0, "grad_norm": 4.285740852355957, "learning_rate": 6.315151515151515e-06, "loss": 0.0084, "num_tokens": 2749384.0, "reward": 0.3990457057952881, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.41745519638061523, "reward_meter_std": 0.3040141463279724, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.3067740499973297, "reward_total_composite_mean": 0.3990457057952881, "reward_total_composite_std": 0.3067740499973297, "reward_total_mean": 0.3990457057952881, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.41745519638061523, "rewards/meter/std": 0.3040141463279724, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.3990457057952881, "rewards/total_composite/std": 0.3067740499973297, "sampling/importance_sampling_ratio/max": 1.500490427017212, "sampling/importance_sampling_ratio/mean": 1.002768635749817, "sampling/importance_sampling_ratio/min": 0.24453596770763397, "sampling/sampling_logp_difference/max": 1.4083929061889648, "sampling/sampling_logp_difference/mean": 0.03294277563691139, "step": 1217 }, { "clip_ratio/high_max": 0.003968254197388887, "clip_ratio/high_mean": 0.003968254197388887, "clip_ratio/low_mean": 0.003937252098694444, "clip_ratio/low_min": 0.003937252098694444, "clip_ratio/region_mean": 0.007905506296083331, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.375, "completions/mean_terminated_length": 64.375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.048549097031354904, "epoch": 0.04892155681407399, "frac_reward_zero_std": 0.0, "grad_norm": 4.59775972366333, "learning_rate": 6.3121212121212125e-06, "loss": -0.0029, "num_tokens": 2751115.0, "reward": 0.9942485094070435, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942485094070435, "reward_meter_std": 0.003570063039660454, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0035700735170394182, "reward_total_composite_mean": 0.9942485094070435, "reward_total_composite_std": 0.003570063039660454, "reward_total_mean": 0.9942485094070435, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942485094070435, "rewards/meter/std": 0.003570063039660454, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942485094070435, "rewards/total_composite/std": 0.003570063039660454, "sampling/importance_sampling_ratio/max": 1.7857288122177124, "sampling/importance_sampling_ratio/mean": 1.001212239265442, "sampling/importance_sampling_ratio/min": 0.19999246299266815, "sampling/sampling_logp_difference/max": 1.6094756126403809, "sampling/sampling_logp_difference/mean": 0.014056126587092876, "step": 1218 }, { "clip_ratio/high_max": 0.023660059785470366, "clip_ratio/high_mean": 0.023660059785470366, "clip_ratio/low_mean": 0.008117978577502072, "clip_ratio/low_min": 0.008117978577502072, "clip_ratio/region_mean": 0.03177803836297244, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 78.375, "completions/mean_terminated_length": 78.375, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.09925029519945383, "epoch": 0.04896172229585894, "frac_reward_zero_std": 0.0, "grad_norm": 4.931517124176025, "learning_rate": 6.309090909090909e-06, "loss": -0.0049, "num_tokens": 2753014.0, "reward": 0.8663658499717712, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9358141422271729, "reward_meter_std": 0.024736735969781876, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.1063641905784607, "reward_total_composite_mean": 0.8663658499717712, "reward_total_composite_std": 0.10636419802904129, "reward_total_mean": 0.8663658499717712, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9358141422271729, "rewards/meter/std": 0.024736735969781876, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8663658499717712, "rewards/total_composite/std": 0.10636419802904129, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9951846599578857, "sampling/importance_sampling_ratio/min": 0.05342717468738556, "sampling/sampling_logp_difference/max": 2.9294357299804688, "sampling/sampling_logp_difference/mean": 0.03355337679386139, "step": 1219 }, { "clip_ratio/high_max": 0.0033783784601837397, "clip_ratio/high_mean": 0.0033783784601837397, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/region_mean": 0.006756756920367479, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 37.0, "completions/mean_terminated_length": 37.0, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.03344483021646738, "epoch": 0.049001887777643895, "frac_reward_zero_std": 0.0, "grad_norm": 0.8351545929908752, "learning_rate": 6.306060606060607e-06, "loss": 0.0002, "num_tokens": 2754558.0, "reward": 0.9974673390388489, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974673390388489, "reward_meter_std": 1.9203745978302322e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.9203745978302322e-05, "reward_total_composite_mean": 0.9974673390388489, "reward_total_composite_std": 1.9203745978302322e-05, "reward_total_mean": 0.9974673390388489, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974673390388489, "rewards/meter/std": 1.9203745978302322e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974673390388489, "rewards/total_composite/std": 1.9203745978302322e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0021648406982422, "sampling/importance_sampling_ratio/min": 0.39682865142822266, "sampling/sampling_logp_difference/max": 0.9242507219314575, "sampling/sampling_logp_difference/mean": 0.010439498350024223, "step": 1220 }, { "clip_ratio/high_max": 0.00996978604234755, "clip_ratio/high_mean": 0.00996978604234755, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/region_mean": 0.013394443551078439, "completions/clipped_ratio": 0.0, "completions/max_length": 219.0, "completions/max_terminated_length": 219.0, "completions/mean_length": 174.625, "completions/mean_terminated_length": 174.625, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.06981830345466733, "epoch": 0.04904205325942885, "frac_reward_zero_std": 0.0, "grad_norm": 3.4342517852783203, "learning_rate": 6.303030303030303e-06, "loss": 0.0928, "num_tokens": 2757419.0, "reward": 0.7635921239852905, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.9972714185714722, "reward_meter_std": 0.0009564639185555279, "reward_repeat_penalty_mean": 0.8042929172515869, "reward_repeat_penalty_std": 0.05763205140829086, "reward_std": 0.1048755943775177, "reward_total_composite_mean": 0.7635921239852905, "reward_total_composite_std": 0.1048756018280983, "reward_total_mean": 0.7635921239852905, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.9972714185714722, "rewards/meter/std": 0.0009564639185555279, "rewards/repeat_penalty/mean": 0.8042929172515869, "rewards/repeat_penalty/std": 0.05763205140829086, "rewards/total_composite/mean": 0.7635921239852905, "rewards/total_composite/std": 0.1048756018280983, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9992591738700867, "sampling/importance_sampling_ratio/min": 0.007855957373976707, "sampling/sampling_logp_difference/max": 4.84648323059082, "sampling/sampling_logp_difference/mean": 0.025953466072678566, "step": 1221 }, { "clip_ratio/high_max": 0.006255614600377157, "clip_ratio/high_mean": 0.006255614600377157, "clip_ratio/low_mean": 0.0027206970553379506, "clip_ratio/low_min": 0.0027206970553379506, "clip_ratio/region_mean": 0.008976311655715108, "completions/clipped_ratio": 0.0, "completions/max_length": 286.0, "completions/max_terminated_length": 286.0, "completions/mean_length": 276.25, "completions/mean_terminated_length": 276.25, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.04638162930496037, "epoch": 0.0490822187412138, "frac_reward_zero_std": 0.0, "grad_norm": 1.311091661453247, "learning_rate": 6.300000000000001e-06, "loss": -0.0077, "num_tokens": 2761573.0, "reward": 0.4972769021987915, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943939447402954, "reward_meter_std": 0.001025979407131672, "reward_repeat_penalty_mean": 0.5, "reward_repeat_penalty_std": 0.15936382114887238, "reward_std": 0.1586521863937378, "reward_total_composite_mean": 0.4972769021987915, "reward_total_composite_std": 0.158652201294899, "reward_total_mean": 0.4972769021987915, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943939447402954, "rewards/meter/std": 0.001025979407131672, "rewards/repeat_penalty/mean": 0.5, "rewards/repeat_penalty/std": 0.15936382114887238, "rewards/total_composite/mean": 0.4972769021987915, "rewards/total_composite/std": 0.158652201294899, "sampling/importance_sampling_ratio/max": 1.5534347295761108, "sampling/importance_sampling_ratio/mean": 0.9976973533630371, "sampling/importance_sampling_ratio/min": 0.03894663602113724, "sampling/sampling_logp_difference/max": 3.245562791824341, "sampling/sampling_logp_difference/mean": 0.013553553260862827, "step": 1222 }, { "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.007247899193316698, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 34.25, "completions/mean_terminated_length": 34.25, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.043639433570206165, "epoch": 0.049122384222998756, "frac_reward_zero_std": 0.0, "grad_norm": 2.368471145629883, "learning_rate": 6.296969696969697e-06, "loss": -0.003, "num_tokens": 2763119.0, "reward": 0.9942933320999146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942933320999146, "reward_meter_std": 8.478895324515179e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.477975643472746e-05, "reward_total_composite_mean": 0.9942933320999146, "reward_total_composite_std": 8.478895324515179e-05, "reward_total_mean": 0.9942933320999146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942933320999146, "rewards/meter/std": 8.478895324515179e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942933320999146, "rewards/total_composite/std": 8.478895324515179e-05, "sampling/importance_sampling_ratio/max": 1.5068020820617676, "sampling/importance_sampling_ratio/mean": 1.0032249689102173, "sampling/importance_sampling_ratio/min": 0.5737356543540955, "sampling/sampling_logp_difference/max": 0.555586576461792, "sampling/sampling_logp_difference/mean": 0.009194603189826012, "step": 1223 }, { "clip_ratio/high_max": 0.016601469949819148, "clip_ratio/high_mean": 0.016601469949819148, "clip_ratio/low_mean": 0.006672008661553264, "clip_ratio/low_min": 0.006672008661553264, "clip_ratio/region_mean": 0.02327347861137241, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 75.375, "completions/mean_terminated_length": 75.375, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.141799321398139, "epoch": 0.04916254970478371, "frac_reward_zero_std": 0.0, "grad_norm": 2.613860607147217, "learning_rate": 6.293939393939394e-06, "loss": 0.0036, "num_tokens": 2764882.0, "reward": 0.8708760142326355, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953185319900513, "reward_meter_std": 0.0019400938181206584, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.1715732365846634, "reward_total_composite_mean": 0.8708760142326355, "reward_total_composite_std": 0.17157325148582458, "reward_total_mean": 0.8708760142326355, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953185319900513, "rewards/meter/std": 0.0019400938181206584, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.8708760142326355, "rewards/total_composite/std": 0.17157325148582458, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003928542137146, "sampling/importance_sampling_ratio/min": 0.06460478901863098, "sampling/sampling_logp_difference/max": 2.739466667175293, "sampling/sampling_logp_difference/mean": 0.028062716126441956, "step": 1224 }, { "clip_ratio/high_max": 0.0030487803742289543, "clip_ratio/high_mean": 0.0030487803742289543, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0030487803742289543, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 122.125, "completions/mean_terminated_length": 122.125, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.018890328239649534, "epoch": 0.049202715186568664, "frac_reward_zero_std": 0.0, "grad_norm": 0.7861045002937317, "learning_rate": 6.290909090909092e-06, "loss": -0.0013, "num_tokens": 2767323.0, "reward": 0.277726948261261, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9720443487167358, "reward_meter_std": 0.0028714225627481937, "reward_repeat_penalty_mean": 0.2857142984867096, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008204064215533435, "reward_total_composite_mean": 0.277726948261261, "reward_total_composite_std": 0.000820409506559372, "reward_total_mean": 0.277726948261261, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9720443487167358, "rewards/meter/std": 0.0028714225627481937, "rewards/repeat_penalty/mean": 0.2857142984867096, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.277726948261261, "rewards/total_composite/std": 0.000820409506559372, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9990015625953674, "sampling/importance_sampling_ratio/min": 0.12622418999671936, "sampling/sampling_logp_difference/max": 2.7687785625457764, "sampling/sampling_logp_difference/mean": 0.009138394147157669, "step": 1225 }, { "clip_ratio/high_max": 0.020103482995182276, "clip_ratio/high_mean": 0.020103482995182276, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/region_mean": 0.024413827806711197, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.25, "completions/mean_terminated_length": 62.25, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.10299600008875132, "epoch": 0.04924288066835362, "frac_reward_zero_std": 0.0, "grad_norm": 7.526054382324219, "learning_rate": 6.287878787878788e-06, "loss": -0.0158, "num_tokens": 2768989.0, "reward": 0.6277604103088379, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8702743053436279, "reward_meter_std": 0.2847423553466797, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.21942584216594696, "reward_total_composite_mean": 0.6277604103088379, "reward_total_composite_std": 0.21942587196826935, "reward_total_mean": 0.6277604103088379, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8702743053436279, "rewards/meter/std": 0.2847423553466797, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.6277604103088379, "rewards/total_composite/std": 0.21942587196826935, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007242202758789, "sampling/importance_sampling_ratio/min": 0.273834228515625, "sampling/sampling_logp_difference/max": 1.2952322959899902, "sampling/sampling_logp_difference/mean": 0.027990855276584625, "step": 1226 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.00390625, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.875, "completions/mean_terminated_length": 32.875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.03647553990595043, "epoch": 0.04928304615013857, "frac_reward_zero_std": 0.0, "grad_norm": 1.5801409482955933, "learning_rate": 6.284848484848486e-06, "loss": -0.0048, "num_tokens": 2770476.0, "reward": 0.9896509051322937, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9896509051322937, "reward_meter_std": 0.00014116229431238025, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00014115417434368283, "reward_total_composite_mean": 0.9896509051322937, "reward_total_composite_std": 0.00014116229431238025, "reward_total_mean": 0.9896509051322937, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9896509051322937, "rewards/meter/std": 0.00014116229431238025, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9896509051322937, "rewards/total_composite/std": 0.00014116229431238025, "sampling/importance_sampling_ratio/max": 1.3669874668121338, "sampling/importance_sampling_ratio/mean": 0.9984188675880432, "sampling/importance_sampling_ratio/min": 0.19587527215480804, "sampling/sampling_logp_difference/max": 1.630277156829834, "sampling/sampling_logp_difference/mean": 0.016333717852830887, "step": 1227 }, { "clip_ratio/high_max": 0.009555137949064374, "clip_ratio/high_mean": 0.009555137949064374, "clip_ratio/low_mean": 0.004665242100600153, "clip_ratio/low_min": 0.004665242100600153, "clip_ratio/region_mean": 0.014220380049664527, "completions/clipped_ratio": 0.0, "completions/max_length": 135.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 131.625, "completions/mean_terminated_length": 131.625, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.0591668093111366, "epoch": 0.049323211631923526, "frac_reward_zero_std": 0.0, "grad_norm": 3.883788824081421, "learning_rate": 6.2818181818181825e-06, "loss": 0.0107, "num_tokens": 2772825.0, "reward": 0.8014565706253052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974247217178345, "reward_meter_std": 0.001207518856972456, "reward_repeat_penalty_mean": 0.8035714626312256, "reward_repeat_penalty_std": 0.10628911107778549, "reward_std": 0.10568942874670029, "reward_total_composite_mean": 0.8014565706253052, "reward_total_composite_std": 0.10568944364786148, "reward_total_mean": 0.8014565706253052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974247217178345, "rewards/meter/std": 0.001207518856972456, "rewards/repeat_penalty/mean": 0.8035714626312256, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.8014565706253052, "rewards/total_composite/std": 0.10568944364786148, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0004632472991943, "sampling/importance_sampling_ratio/min": 0.04599026218056679, "sampling/sampling_logp_difference/max": 3.0793256759643555, "sampling/sampling_logp_difference/mean": 0.01869257725775242, "step": 1228 }, { "clip_ratio/high_max": 0.008696309756487608, "clip_ratio/high_mean": 0.008696309756487608, "clip_ratio/low_mean": 0.006359649356454611, "clip_ratio/low_min": 0.006359649356454611, "clip_ratio/region_mean": 0.015055959112942219, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 57.5, "completions/mean_terminated_length": 57.5, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.15489636361598969, "epoch": 0.04936337711370848, "frac_reward_zero_std": 0.0, "grad_norm": 6.317329406738281, "learning_rate": 6.27878787878788e-06, "loss": 0.0111, "num_tokens": 2774629.0, "reward": 0.11471378803253174, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.11471378803253174, "reward_meter_std": 0.10604586452245712, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.10604586452245712, "reward_total_composite_mean": 0.11471378803253174, "reward_total_composite_std": 0.10604586452245712, "reward_total_mean": 0.11471378803253174, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.11471378803253174, "rewards/meter/std": 0.10604586452245712, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.11471378803253174, "rewards/total_composite/std": 0.10604586452245712, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000756025314331, "sampling/importance_sampling_ratio/min": 0.2904861271381378, "sampling/sampling_logp_difference/max": 1.2361993789672852, "sampling/sampling_logp_difference/mean": 0.033396147191524506, "step": 1229 }, { "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/low_mean": 0.007464349502697587, "clip_ratio/low_min": 0.007464349502697587, "clip_ratio/region_mean": 0.011140820104628801, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.5, "completions/mean_terminated_length": 33.5, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.07495713606476784, "epoch": 0.049403542595493434, "frac_reward_zero_std": 0.0, "grad_norm": 3.3790931701660156, "learning_rate": 6.275757575757576e-06, "loss": 0.0035, "num_tokens": 2776041.0, "reward": 0.9702644944190979, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9702644944190979, "reward_meter_std": 0.0008763103978708386, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008763210498727858, "reward_total_composite_mean": 0.9702644944190979, "reward_total_composite_std": 0.0008763103978708386, "reward_total_mean": 0.9702644944190979, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9702644944190979, "rewards/meter/std": 0.0008763103978708386, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9702644944190979, "rewards/total_composite/std": 0.0008763103978708386, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0119966268539429, "sampling/importance_sampling_ratio/min": 0.3646947145462036, "sampling/sampling_logp_difference/max": 1.2655248641967773, "sampling/sampling_logp_difference/mean": 0.022185776382684708, "step": 1230 }, { "clip_ratio/high_max": 0.03711274731904268, "clip_ratio/high_mean": 0.03711274731904268, "clip_ratio/low_mean": 0.028201664797961712, "clip_ratio/low_min": 0.028201664797961712, "clip_ratio/region_mean": 0.0653144121170044, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 58.5, "completions/mean_terminated_length": 58.5, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.5035125408321619, "epoch": 0.04944370807727839, "frac_reward_zero_std": 0.0, "grad_norm": 6.076712608337402, "learning_rate": 6.2727272727272734e-06, "loss": -0.0146, "num_tokens": 2777725.0, "reward": 0.6401726603507996, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6401726603507996, "reward_meter_std": 0.34182804822921753, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.34182801842689514, "reward_total_composite_mean": 0.6401726603507996, "reward_total_composite_std": 0.34182804822921753, "reward_total_mean": 0.6401726603507996, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6401726603507996, "rewards/meter/std": 0.34182804822921753, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6401726603507996, "rewards/total_composite/std": 0.34182804822921753, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0154814720153809, "sampling/importance_sampling_ratio/min": 0.24655072391033173, "sampling/sampling_logp_difference/max": 1.4318437576293945, "sampling/sampling_logp_difference/mean": 0.06882074475288391, "step": 1231 }, { "clip_ratio/high_max": 0.01965810963883996, "clip_ratio/high_mean": 0.01965810963883996, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.01965810963883996, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.25, "completions/mean_terminated_length": 76.25, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.1252383915707469, "epoch": 0.04948387355906334, "frac_reward_zero_std": 0.0, "grad_norm": 4.830690860748291, "learning_rate": 6.26969696969697e-06, "loss": -0.0004, "num_tokens": 2779575.0, "reward": 0.8763631582260132, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8763631582260132, "reward_meter_std": 0.33702343702316284, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.33702343702316284, "reward_total_composite_mean": 0.8763631582260132, "reward_total_composite_std": 0.33702343702316284, "reward_total_mean": 0.8763631582260132, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8763631582260132, "rewards/meter/std": 0.33702343702316284, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8763631582260132, "rewards/total_composite/std": 0.33702343702316284, "sampling/importance_sampling_ratio/max": 1.7376562356948853, "sampling/importance_sampling_ratio/mean": 0.9978901743888855, "sampling/importance_sampling_ratio/min": 0.021180974319577217, "sampling/sampling_logp_difference/max": 3.854651927947998, "sampling/sampling_logp_difference/mean": 0.03446757048368454, "step": 1232 }, { "clip_ratio/high_max": 0.006812343490310013, "clip_ratio/high_mean": 0.006812343490310013, "clip_ratio/low_mean": 0.004945102846249938, "clip_ratio/low_min": 0.004945102846249938, "clip_ratio/region_mean": 0.011757446336559951, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 150.125, "completions/mean_terminated_length": 150.125, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.08290160167962313, "epoch": 0.049524039040848296, "frac_reward_zero_std": 0.0, "grad_norm": 2.373234748840332, "learning_rate": 6.266666666666668e-06, "loss": 0.0154, "num_tokens": 2782216.0, "reward": 0.6755450963973999, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9954185485839844, "reward_meter_std": 0.0016214650822803378, "reward_repeat_penalty_mean": 0.6785714626312256, "reward_repeat_penalty_std": 0.14787118136882782, "reward_std": 0.1476629078388214, "reward_total_composite_mean": 0.6755450963973999, "reward_total_composite_std": 0.1476629376411438, "reward_total_mean": 0.6755450963973999, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9954185485839844, "rewards/meter/std": 0.0016214650822803378, "rewards/repeat_penalty/mean": 0.6785714626312256, "rewards/repeat_penalty/std": 0.14787118136882782, "rewards/total_composite/mean": 0.6755450963973999, "rewards/total_composite/std": 0.1476629376411438, "sampling/importance_sampling_ratio/max": 1.8110910654067993, "sampling/importance_sampling_ratio/mean": 1.002487301826477, "sampling/importance_sampling_ratio/min": 0.34533634781837463, "sampling/sampling_logp_difference/max": 1.0632364749908447, "sampling/sampling_logp_difference/mean": 0.013783158734440804, "step": 1233 }, { "clip_ratio/high_max": 0.00862484163371846, "clip_ratio/high_mean": 0.00862484163371846, "clip_ratio/low_mean": 0.005789120797999203, "clip_ratio/low_min": 0.005789120797999203, "clip_ratio/region_mean": 0.014413962431717664, "completions/clipped_ratio": 0.0, "completions/max_length": 260.0, "completions/max_terminated_length": 260.0, "completions/mean_length": 242.875, "completions/mean_terminated_length": 242.875, "completions/min_length": 227.0, "completions/min_terminated_length": 227.0, "entropy": 0.09457159182056785, "epoch": 0.04956420452263325, "frac_reward_zero_std": 0.0, "grad_norm": 1.7159820795059204, "learning_rate": 6.263636363636364e-06, "loss": -0.002, "num_tokens": 2785815.0, "reward": 0.4686053991317749, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.737500011920929, "reward_count_adherence_std": 0.05175492912530899, "reward_meter_mean": 0.9932608008384705, "reward_meter_std": 0.003682313719764352, "reward_repeat_penalty_mean": 0.6418956518173218, "reward_repeat_penalty_std": 0.06913584470748901, "reward_std": 0.04152139648795128, "reward_total_composite_mean": 0.4686053991317749, "reward_total_composite_std": 0.04152140021324158, "reward_total_mean": 0.4686053991317749, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.737500011920929, "rewards/count_adherence/std": 0.05175492912530899, "rewards/meter/mean": 0.9932608008384705, "rewards/meter/std": 0.003682313719764352, "rewards/repeat_penalty/mean": 0.6418956518173218, "rewards/repeat_penalty/std": 0.06913584470748901, "rewards/total_composite/mean": 0.4686053991317749, "rewards/total_composite/std": 0.04152140021324158, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0014477968215942, "sampling/importance_sampling_ratio/min": 0.030284356325864792, "sampling/sampling_logp_difference/max": 3.497123956680298, "sampling/sampling_logp_difference/mean": 0.022554513067007065, "step": 1234 }, { "clip_ratio/high_max": 0.031061405315995216, "clip_ratio/high_mean": 0.031061405315995216, "clip_ratio/low_mean": 0.006172839552164078, "clip_ratio/low_min": 0.006172839552164078, "clip_ratio/region_mean": 0.037234244868159294, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 80.5, "completions/mean_terminated_length": 80.5, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.10135750938206911, "epoch": 0.049604370004418204, "frac_reward_zero_std": 0.0, "grad_norm": 15.415648460388184, "learning_rate": 6.260606060606062e-06, "loss": 0.011, "num_tokens": 2787651.0, "reward": 0.9315690398216248, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9555257558822632, "reward_meter_std": 0.013856525532901287, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06806813925504684, "reward_total_composite_mean": 0.9315690398216248, "reward_total_composite_std": 0.06806813180446625, "reward_total_mean": 0.9315690398216248, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9555257558822632, "rewards/meter/std": 0.013856525532901287, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9315690398216248, "rewards/total_composite/std": 0.06806813180446625, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9921613335609436, "sampling/importance_sampling_ratio/min": 0.003756270743906498, "sampling/sampling_logp_difference/max": 5.584328651428223, "sampling/sampling_logp_difference/mean": 0.04006582126021385, "step": 1235 }, { "clip_ratio/high_max": 0.006200521253049374, "clip_ratio/high_mean": 0.006200521253049374, "clip_ratio/low_mean": 0.002725058700889349, "clip_ratio/low_min": 0.002725058700889349, "clip_ratio/region_mean": 0.008925579953938723, "completions/clipped_ratio": 0.0, "completions/max_length": 186.0, "completions/max_terminated_length": 186.0, "completions/mean_length": 181.625, "completions/mean_terminated_length": 181.625, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.050744547275826335, "epoch": 0.04964453548620316, "frac_reward_zero_std": 0.0, "grad_norm": 1.9874340295791626, "learning_rate": 6.257575757575758e-06, "loss": 0.0094, "num_tokens": 2790696.0, "reward": 0.5256986618041992, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979325532913208, "reward_meter_std": 0.00011955534137086943, "reward_repeat_penalty_mean": 0.737500011920929, "reward_repeat_penalty_std": 0.07440238445997238, "reward_std": 0.05306261032819748, "reward_total_composite_mean": 0.5256986618041992, "reward_total_composite_std": 0.053062621504068375, "reward_total_mean": 0.5256986618041992, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979325532913208, "rewards/meter/std": 0.00011955534137086943, "rewards/repeat_penalty/mean": 0.737500011920929, "rewards/repeat_penalty/std": 0.07440238445997238, "rewards/total_composite/mean": 0.5256986618041992, "rewards/total_composite/std": 0.053062621504068375, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016508102416992, "sampling/importance_sampling_ratio/min": 0.22438688576221466, "sampling/sampling_logp_difference/max": 1.4943835735321045, "sampling/sampling_logp_difference/mean": 0.012886044569313526, "step": 1236 }, { "clip_ratio/high_max": 0.028940699994564056, "clip_ratio/high_mean": 0.028940699994564056, "clip_ratio/low_mean": 0.03385214158333838, "clip_ratio/low_min": 0.03385214158333838, "clip_ratio/region_mean": 0.06279284157790244, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.375, "completions/mean_terminated_length": 67.375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.3711269237101078, "epoch": 0.04968470096798811, "frac_reward_zero_std": 0.0, "grad_norm": 6.145448684692383, "learning_rate": 6.254545454545455e-06, "loss": 0.0102, "num_tokens": 2792675.0, "reward": 0.35225385427474976, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.35231250524520874, "reward_meter_std": 0.3081033229827881, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.30817967653274536, "reward_total_composite_mean": 0.35225385427474976, "reward_total_composite_std": 0.30817967653274536, "reward_total_mean": 0.35225385427474976, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.35231250524520874, "rewards/meter/std": 0.3081033229827881, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.35225385427474976, "rewards/total_composite/std": 0.30817967653274536, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002850890159607, "sampling/importance_sampling_ratio/min": 0.037932947278022766, "sampling/sampling_logp_difference/max": 3.271935224533081, "sampling/sampling_logp_difference/mean": 0.0802878737449646, "step": 1237 }, { "clip_ratio/high_max": 0.017562699620611966, "clip_ratio/high_mean": 0.017562699620611966, "clip_ratio/low_mean": 0.009122495306655765, "clip_ratio/low_min": 0.009122495306655765, "clip_ratio/region_mean": 0.02668519492726773, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 85.625, "completions/mean_terminated_length": 85.625, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.1898734886199236, "epoch": 0.049724866449773066, "frac_reward_zero_std": 0.0, "grad_norm": 6.303664684295654, "learning_rate": 6.251515151515152e-06, "loss": -0.0256, "num_tokens": 2794768.0, "reward": 0.452033668756485, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.15430334210395813, "reward_meter_mean": 0.6144058704376221, "reward_meter_std": 0.3429763913154602, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.2605564594268799, "reward_total_composite_mean": 0.452033668756485, "reward_total_composite_std": 0.2605564594268799, "reward_total_mean": 0.452033668756485, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.15430334210395813, "rewards/meter/mean": 0.6144058704376221, "rewards/meter/std": 0.3429763913154602, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.452033668756485, "rewards/total_composite/std": 0.2605564594268799, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0040868520736694, "sampling/importance_sampling_ratio/min": 0.09410162270069122, "sampling/sampling_logp_difference/max": 2.363379955291748, "sampling/sampling_logp_difference/mean": 0.05436992645263672, "step": 1238 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.007464349502697587, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 34.0, "completions/mean_terminated_length": 34.0, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.04970884299837053, "epoch": 0.04976503193155802, "frac_reward_zero_std": 0.0, "grad_norm": 11.653518676757812, "learning_rate": 6.248484848484849e-06, "loss": 0.0011, "num_tokens": 2796176.0, "reward": 0.9623700380325317, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9623700380325317, "reward_meter_std": 0.017837192863225937, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.017837194725871086, "reward_total_composite_mean": 0.9623700380325317, "reward_total_composite_std": 0.017837192863225937, "reward_total_mean": 0.9623700380325317, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9623700380325317, "rewards/meter/std": 0.017837192863225937, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9623700380325317, "rewards/total_composite/std": 0.017837192863225937, "sampling/importance_sampling_ratio/max": 1.2581915855407715, "sampling/importance_sampling_ratio/mean": 0.9969296455383301, "sampling/importance_sampling_ratio/min": 0.21284890174865723, "sampling/sampling_logp_difference/max": 1.5471727848052979, "sampling/sampling_logp_difference/mean": 0.017896274104714394, "step": 1239 }, { "clip_ratio/high_max": 0.037521140300668776, "clip_ratio/high_mean": 0.037521140300668776, "clip_ratio/low_mean": 0.023755351547151804, "clip_ratio/low_min": 0.023755351547151804, "clip_ratio/region_mean": 0.06127649184782058, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.17542924359440804, "epoch": 0.04980519741334297, "frac_reward_zero_std": 0.0, "grad_norm": 18.420658111572266, "learning_rate": 6.245454545454545e-06, "loss": -0.0239, "num_tokens": 2797992.0, "reward": 0.750508189201355, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.750508189201355, "reward_meter_std": 0.30968019366264343, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3096802234649658, "reward_total_composite_mean": 0.750508189201355, "reward_total_composite_std": 0.30968019366264343, "reward_total_mean": 0.750508189201355, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.750508189201355, "rewards/meter/std": 0.30968019366264343, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.750508189201355, "rewards/total_composite/std": 0.30968019366264343, "sampling/importance_sampling_ratio/max": 1.6765276193618774, "sampling/importance_sampling_ratio/mean": 0.9967520833015442, "sampling/importance_sampling_ratio/min": 0.07774890959262848, "sampling/sampling_logp_difference/max": 2.5542707443237305, "sampling/sampling_logp_difference/mean": 0.052898943424224854, "step": 1240 }, { "clip_ratio/high_max": 0.030546388239599764, "clip_ratio/high_mean": 0.030546388239599764, "clip_ratio/low_mean": 0.007147361640818417, "clip_ratio/low_min": 0.007147361640818417, "clip_ratio/region_mean": 0.03769374988041818, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 69.75, "completions/mean_terminated_length": 69.75, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.134712896309793, "epoch": 0.04984536289512793, "frac_reward_zero_std": 0.0, "grad_norm": 5.958988189697266, "learning_rate": 6.2424242424242434e-06, "loss": 0.0106, "num_tokens": 2799886.0, "reward": 0.9957868456840515, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957868456840515, "reward_meter_std": 0.0033736079931259155, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0033736105542629957, "reward_total_composite_mean": 0.9957868456840515, "reward_total_composite_std": 0.0033736079931259155, "reward_total_mean": 0.9957868456840515, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957868456840515, "rewards/meter/std": 0.0033736079931259155, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957868456840515, "rewards/total_composite/std": 0.0033736079931259155, "sampling/importance_sampling_ratio/max": 1.7876514196395874, "sampling/importance_sampling_ratio/mean": 0.9955186247825623, "sampling/importance_sampling_ratio/min": 0.2759105861186981, "sampling/sampling_logp_difference/max": 1.2876784801483154, "sampling/sampling_logp_difference/mean": 0.0355050228536129, "step": 1241 }, { "clip_ratio/high_max": 0.024632065324112773, "clip_ratio/high_mean": 0.024632065324112773, "clip_ratio/low_mean": 0.006905802641995251, "clip_ratio/low_min": 0.006905802641995251, "clip_ratio/region_mean": 0.031537867966108024, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.75, "completions/mean_terminated_length": 70.75, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.14225120842456818, "epoch": 0.04988552837691288, "frac_reward_zero_std": 0.0, "grad_norm": 16.907014846801758, "learning_rate": 6.23939393939394e-06, "loss": 0.0265, "num_tokens": 2801684.0, "reward": 0.9978107810020447, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978107810020447, "reward_meter_std": 0.0009332987247034907, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009333047200925648, "reward_total_composite_mean": 0.9978107810020447, "reward_total_composite_std": 0.0009332987247034907, "reward_total_mean": 0.9978107810020447, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978107810020447, "rewards/meter/std": 0.0009332987247034907, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978107810020447, "rewards/total_composite/std": 0.0009332987247034907, "sampling/importance_sampling_ratio/max": 1.5690217018127441, "sampling/importance_sampling_ratio/mean": 1.003254771232605, "sampling/importance_sampling_ratio/min": 0.07646137475967407, "sampling/sampling_logp_difference/max": 2.570969581604004, "sampling/sampling_logp_difference/mean": 0.03522227331995964, "step": 1242 }, { "clip_ratio/high_max": 0.017581941094249487, "clip_ratio/high_mean": 0.017581941094249487, "clip_ratio/low_mean": 0.021577381063252687, "clip_ratio/low_min": 0.021577381063252687, "clip_ratio/region_mean": 0.039159322157502174, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 63.875, "completions/mean_terminated_length": 63.875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.10157017782330513, "epoch": 0.049925693858697835, "frac_reward_zero_std": 0.0, "grad_norm": 8.487194061279297, "learning_rate": 6.236363636363637e-06, "loss": 0.0058, "num_tokens": 2803443.0, "reward": 0.9128004312515259, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9128004312515259, "reward_meter_std": 0.03738430514931679, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.037384290248155594, "reward_total_composite_mean": 0.9128004312515259, "reward_total_composite_std": 0.03738430514931679, "reward_total_mean": 0.9128004312515259, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9128004312515259, "rewards/meter/std": 0.03738430514931679, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9128004312515259, "rewards/total_composite/std": 0.03738430514931679, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0069886445999146, "sampling/importance_sampling_ratio/min": 0.002972492016851902, "sampling/sampling_logp_difference/max": 5.818354606628418, "sampling/sampling_logp_difference/mean": 0.04461809620261192, "step": 1243 }, { "clip_ratio/high_max": 0.009529390663374215, "clip_ratio/high_mean": 0.009529390663374215, "clip_ratio/low_mean": 0.009928334911819547, "clip_ratio/low_min": 0.009928334911819547, "clip_ratio/region_mean": 0.019457725575193763, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 128.125, "completions/mean_terminated_length": 128.125, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.1307503404095769, "epoch": 0.04996585934048279, "frac_reward_zero_std": 0.0, "grad_norm": 2.8616859912872314, "learning_rate": 6.2333333333333335e-06, "loss": -0.0386, "num_tokens": 2805756.0, "reward": 0.7401801347732544, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9177215695381165, "reward_meter_std": 0.08298756927251816, "reward_repeat_penalty_mean": 0.8363094925880432, "reward_repeat_penalty_std": 0.09127599745988846, "reward_std": 0.10087206959724426, "reward_total_composite_mean": 0.7401801347732544, "reward_total_composite_std": 0.10087206214666367, "reward_total_mean": 0.7401801347732544, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9177215695381165, "rewards/meter/std": 0.08298756927251816, "rewards/repeat_penalty/mean": 0.8363094925880432, "rewards/repeat_penalty/std": 0.09127599745988846, "rewards/total_composite/mean": 0.7401801347732544, "rewards/total_composite/std": 0.10087206214666367, "sampling/importance_sampling_ratio/max": 1.7540781497955322, "sampling/importance_sampling_ratio/mean": 1.0052604675292969, "sampling/importance_sampling_ratio/min": 0.46336889266967773, "sampling/sampling_logp_difference/max": 0.7692317962646484, "sampling/sampling_logp_difference/mean": 0.021584779024124146, "step": 1244 }, { "clip_ratio/high_max": 0.004898735671304166, "clip_ratio/high_mean": 0.004898735671304166, "clip_ratio/low_mean": 0.009747638367116451, "clip_ratio/low_min": 0.009747638367116451, "clip_ratio/region_mean": 0.014646374038420618, "completions/clipped_ratio": 0.0, "completions/max_length": 239.0, "completions/max_terminated_length": 239.0, "completions/mean_length": 230.0, "completions/mean_terminated_length": 230.0, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.06460519693791866, "epoch": 0.05000602482226774, "frac_reward_zero_std": 0.0, "grad_norm": 2.1829257011413574, "learning_rate": 6.230303030303031e-06, "loss": 0.0098, "num_tokens": 2809292.0, "reward": 0.5989092588424683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7638888955116272, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.9974883198738098, "reward_meter_std": 0.0011670741951093078, "reward_repeat_penalty_mean": 0.7868589758872986, "reward_repeat_penalty_std": 0.07841575890779495, "reward_std": 0.061690811067819595, "reward_total_composite_mean": 0.5989092588424683, "reward_total_composite_std": 0.061690803617239, "reward_total_mean": 0.5989092588424683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7638888955116272, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.9974883198738098, "rewards/meter/std": 0.0011670741951093078, "rewards/repeat_penalty/mean": 0.7868589758872986, "rewards/repeat_penalty/std": 0.07841575890779495, "rewards/total_composite/mean": 0.5989092588424683, "rewards/total_composite/std": 0.061690803617239, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9981163740158081, "sampling/importance_sampling_ratio/min": 0.10331789404153824, "sampling/sampling_logp_difference/max": 2.269944667816162, "sampling/sampling_logp_difference/mean": 0.01827998273074627, "step": 1245 }, { "clip_ratio/high_max": 0.007221258128993213, "clip_ratio/high_mean": 0.007221258128993213, "clip_ratio/low_mean": 0.007273018010891974, "clip_ratio/low_min": 0.007273018010891974, "clip_ratio/region_mean": 0.014494276139885187, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.875, "completions/mean_terminated_length": 68.875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.08360549993813038, "epoch": 0.0500461903040527, "frac_reward_zero_std": 0.0, "grad_norm": 5.752688407897949, "learning_rate": 6.227272727272727e-06, "loss": -0.0011, "num_tokens": 2811283.0, "reward": 0.7574069499969482, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7574069499969482, "reward_meter_std": 0.31513679027557373, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.31513676047325134, "reward_total_composite_mean": 0.7574069499969482, "reward_total_composite_std": 0.31513679027557373, "reward_total_mean": 0.7574069499969482, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7574069499969482, "rewards/meter/std": 0.31513679027557373, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7574069499969482, "rewards/total_composite/std": 0.31513679027557373, "sampling/importance_sampling_ratio/max": 1.6327474117279053, "sampling/importance_sampling_ratio/mean": 0.9997113347053528, "sampling/importance_sampling_ratio/min": 0.4760926067829132, "sampling/sampling_logp_difference/max": 0.742142915725708, "sampling/sampling_logp_difference/mean": 0.017493808642029762, "step": 1246 }, { "clip_ratio/high_max": 0.007731958641670644, "clip_ratio/high_mean": 0.007731958641670644, "clip_ratio/low_mean": 0.011977351736277342, "clip_ratio/low_min": 0.011977351736277342, "clip_ratio/region_mean": 0.019709310377947986, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 95.5, "completions/mean_terminated_length": 95.5, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.13955960143357515, "epoch": 0.05008635578583765, "frac_reward_zero_std": 0.0, "grad_norm": 3.6555347442626953, "learning_rate": 6.224242424242425e-06, "loss": -0.0245, "num_tokens": 2813271.0, "reward": 0.8177590370178223, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.95280921459198, "reward_meter_std": 0.017720038071274757, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.12049128115177155, "reward_total_composite_mean": 0.8177590370178223, "reward_total_composite_std": 0.12049128115177155, "reward_total_mean": 0.8177590370178223, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.95280921459198, "rewards/meter/std": 0.017720038071274757, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8177590370178223, "rewards/total_composite/std": 0.12049128115177155, "sampling/importance_sampling_ratio/max": 1.4265241622924805, "sampling/importance_sampling_ratio/mean": 1.0025370121002197, "sampling/importance_sampling_ratio/min": 0.5583049654960632, "sampling/sampling_logp_difference/max": 0.5828499794006348, "sampling/sampling_logp_difference/mean": 0.020115824416279793, "step": 1247 }, { "clip_ratio/high_max": 0.009827935369685292, "clip_ratio/high_mean": 0.009827935369685292, "clip_ratio/low_mean": 0.004895833437331021, "clip_ratio/low_min": 0.004895833437331021, "clip_ratio/region_mean": 0.014723768807016313, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 76.75, "completions/mean_terminated_length": 76.75, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.14584199711680412, "epoch": 0.050126521267622605, "frac_reward_zero_std": 0.0, "grad_norm": 3.7371089458465576, "learning_rate": 6.221212121212121e-06, "loss": 0.0024, "num_tokens": 2815189.0, "reward": 0.9958518147468567, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958518147468567, "reward_meter_std": 0.0027126488275825977, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0027126302011311054, "reward_total_composite_mean": 0.9958518147468567, "reward_total_composite_std": 0.0027126488275825977, "reward_total_mean": 0.9958518147468567, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958518147468567, "rewards/meter/std": 0.0027126488275825977, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958518147468567, "rewards/total_composite/std": 0.0027126488275825977, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.999036431312561, "sampling/importance_sampling_ratio/min": 0.019202744588255882, "sampling/sampling_logp_difference/max": 3.952702045440674, "sampling/sampling_logp_difference/mean": 0.03340546786785126, "step": 1248 }, { "clip_ratio/high_max": 0.0036526747280731797, "clip_ratio/high_mean": 0.0036526747280731797, "clip_ratio/low_mean": 0.007488123839721084, "clip_ratio/low_min": 0.007488123839721084, "clip_ratio/region_mean": 0.011140798567794263, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 100.75, "completions/mean_terminated_length": 100.75, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.06938632484525442, "epoch": 0.05016668674940756, "frac_reward_zero_std": 0.0, "grad_norm": 2.240663528442383, "learning_rate": 6.218181818181819e-06, "loss": -0.0042, "num_tokens": 2817451.0, "reward": 0.898347020149231, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981772899627686, "reward_meter_std": 0.0003791229974012822, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.1065896600484848, "reward_total_composite_mean": 0.898347020149231, "reward_total_composite_std": 0.1065896674990654, "reward_total_mean": 0.898347020149231, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981772899627686, "rewards/meter/std": 0.0003791229974012822, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.898347020149231, "rewards/total_composite/std": 0.1065896674990654, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003831148147583, "sampling/importance_sampling_ratio/min": 0.516797661781311, "sampling/sampling_logp_difference/max": 0.8134353160858154, "sampling/sampling_logp_difference/mean": 0.012546716257929802, "step": 1249 }, { "clip_ratio/high_max": 0.01242569915484637, "clip_ratio/high_mean": 0.01242569915484637, "clip_ratio/low_mean": 0.004870129749178886, "clip_ratio/low_min": 0.004870129749178886, "clip_ratio/region_mean": 0.017295828904025257, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 79.875, "completions/mean_terminated_length": 79.875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.09262991230934858, "epoch": 0.05020685223119251, "frac_reward_zero_std": 0.0, "grad_norm": 2.518706798553467, "learning_rate": 6.215151515151515e-06, "loss": -0.0122, "num_tokens": 2819322.0, "reward": 0.9494280815124512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9909250736236572, "reward_meter_std": 0.0028839793521910906, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11536785960197449, "reward_total_composite_mean": 0.9494280815124512, "reward_total_composite_std": 0.11536786705255508, "reward_total_mean": 0.9494280815124512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9909250736236572, "rewards/meter/std": 0.0028839793521910906, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9494280815124512, "rewards/total_composite/std": 0.11536786705255508, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0048216581344604, "sampling/importance_sampling_ratio/min": 0.0684998482465744, "sampling/sampling_logp_difference/max": 2.6809237003326416, "sampling/sampling_logp_difference/mean": 0.0175126064568758, "step": 1250 }, { "epoch": 0.05020685223119251, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 286.84615384615387, "eval_completions/max_terminated_length": 286.84615384615387, "eval_completions/mean_length": 172.58653846153845, "eval_completions/mean_terminated_length": 172.58653846153845, "eval_completions/min_length": 65.3076923076923, "eval_completions/min_terminated_length": 65.3076923076923, "eval_entropy": 0.10610828204796864, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 2819322.0, "eval_reward": 0.47181554253284747, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.8135509353417617, "eval_reward_count_adherence_std": 0.17685549706220627, "eval_reward_meter_mean": 0.689526264484112, "eval_reward_meter_std": 0.4130670749224149, "eval_reward_repeat_penalty_mean": 0.8221937005336468, "eval_reward_repeat_penalty_std": 0.12831746328335542, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.47181554253284747, "eval_reward_total_composite_std": 0.3225762511675174, "eval_reward_total_mean": 0.47181554253284747, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.8135509353417617, "eval_rewards/count_adherence/std": 0.17685549706220627, "eval_rewards/meter/mean": 0.689526264484112, "eval_rewards/meter/std": 0.4130670749224149, "eval_rewards/repeat_penalty/mean": 0.8221937005336468, "eval_rewards/repeat_penalty/std": 0.12831746328335542, "eval_rewards/total_composite/mean": 0.47181554253284747, "eval_rewards/total_composite/std": 0.3225762511675174, "eval_runtime": 56.3598, "eval_samples_per_second": 1.845, "eval_sampling/importance_sampling_ratio/max": 1.3588372377248912, "eval_sampling/importance_sampling_ratio/mean": 1.002467971581679, "eval_sampling/importance_sampling_ratio/min": 0.4489339819321266, "eval_sampling/sampling_logp_difference/max": 0.8327025633591872, "eval_sampling/sampling_logp_difference/mean": 0.012653420428530527, "eval_steps_per_second": 0.231, "step": 1250 }, { "clip_ratio/high_max": 0.01761134958360344, "clip_ratio/high_mean": 0.01761134958360344, "clip_ratio/low_mean": 0.007605619612149894, "clip_ratio/low_min": 0.007605619612149894, "clip_ratio/region_mean": 0.025216969195753336, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 99.25, "completions/mean_terminated_length": 99.25, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.09751791087910533, "epoch": 0.05024701771297747, "frac_reward_zero_std": 0.0, "grad_norm": 6.843512058258057, "learning_rate": 6.212121212121213e-06, "loss": 0.0013, "num_tokens": 2821356.0, "reward": 0.8895110487937927, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9614334106445312, "reward_meter_std": 0.015211360529065132, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10235292464494705, "reward_total_composite_mean": 0.8895110487937927, "reward_total_composite_std": 0.10235293209552765, "reward_total_mean": 0.8895110487937927, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9614334106445312, "rewards/meter/std": 0.015211360529065132, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8895110487937927, "rewards/total_composite/std": 0.10235293209552765, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9989413022994995, "sampling/importance_sampling_ratio/min": 0.018923774361610413, "sampling/sampling_logp_difference/max": 3.9673361778259277, "sampling/sampling_logp_difference/mean": 0.039652157574892044, "step": 1251 }, { "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/low_mean": 0.01109143253415823, "clip_ratio/low_min": 0.01109143253415823, "clip_ratio/region_mean": 0.013642452890053391, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 147.375, "completions/mean_terminated_length": 147.375, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.09728506673127413, "epoch": 0.05028718319476242, "frac_reward_zero_std": 0.0, "grad_norm": 1.9272445440292358, "learning_rate": 6.209090909090909e-06, "loss": 0.0046, "num_tokens": 2824023.0, "reward": 0.7294289469718933, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962889552116394, "reward_meter_std": 0.0008069492760114372, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.050382472574710846, "reward_total_composite_mean": 0.7294289469718933, "reward_total_composite_std": 0.050382453948259354, "reward_total_mean": 0.7294289469718933, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962889552116394, "rewards/meter/std": 0.0008069492760114372, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.7294289469718933, "rewards/total_composite/std": 0.050382453948259354, "sampling/importance_sampling_ratio/max": 1.5003139972686768, "sampling/importance_sampling_ratio/mean": 1.0037672519683838, "sampling/importance_sampling_ratio/min": 0.23551644384860992, "sampling/sampling_logp_difference/max": 1.445974588394165, "sampling/sampling_logp_difference/mean": 0.014175360091030598, "step": 1252 }, { "clip_ratio/high_max": 0.008442942867986858, "clip_ratio/high_mean": 0.008442942867986858, "clip_ratio/low_mean": 0.005023152916692197, "clip_ratio/low_min": 0.005023152916692197, "clip_ratio/region_mean": 0.013466095784679055, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 290.875, "completions/mean_terminated_length": 290.875, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.09894186165183783, "epoch": 0.050327348676547375, "frac_reward_zero_std": 0.0, "grad_norm": 2.051161050796509, "learning_rate": 6.206060606060606e-06, "loss": -0.0071, "num_tokens": 2828134.0, "reward": 0.3865600824356079, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5576923489570618, "reward_count_adherence_std": 0.03560846298933029, "reward_meter_mean": 0.9930562376976013, "reward_meter_std": 0.01025536097586155, "reward_repeat_penalty_mean": 0.6976648569107056, "reward_repeat_penalty_std": 0.03868091478943825, "reward_std": 0.0352591797709465, "reward_total_composite_mean": 0.3865600824356079, "reward_total_composite_std": 0.0352591797709465, "reward_total_mean": 0.3865600824356079, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5576923489570618, "rewards/count_adherence/std": 0.03560846298933029, "rewards/meter/mean": 0.9930562376976013, "rewards/meter/std": 0.01025536097586155, "rewards/repeat_penalty/mean": 0.6976648569107056, "rewards/repeat_penalty/std": 0.03868091478943825, "rewards/total_composite/mean": 0.3865600824356079, "rewards/total_composite/std": 0.0352591797709465, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9969112277030945, "sampling/importance_sampling_ratio/min": 0.11250852793455124, "sampling/sampling_logp_difference/max": 2.5028674602508545, "sampling/sampling_logp_difference/mean": 0.02420906163752079, "step": 1253 }, { "clip_ratio/high_max": 0.00916612590663135, "clip_ratio/high_mean": 0.00916612590663135, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.011031797504983842, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.75, "completions/mean_terminated_length": 67.75, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.052088204305619, "epoch": 0.05036751415833233, "frac_reward_zero_std": 0.0, "grad_norm": 4.324112892150879, "learning_rate": 6.203030303030304e-06, "loss": -0.0038, "num_tokens": 2829964.0, "reward": 0.9599640965461731, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9599640965461731, "reward_meter_std": 0.03301013261079788, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03301013633608818, "reward_total_composite_mean": 0.9599640965461731, "reward_total_composite_std": 0.03301013261079788, "reward_total_mean": 0.9599640965461731, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9599640965461731, "rewards/meter/std": 0.03301013261079788, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9599640965461731, "rewards/total_composite/std": 0.03301013261079788, "sampling/importance_sampling_ratio/max": 1.6136887073516846, "sampling/importance_sampling_ratio/mean": 1.0017342567443848, "sampling/importance_sampling_ratio/min": 0.12102648615837097, "sampling/sampling_logp_difference/max": 2.111745834350586, "sampling/sampling_logp_difference/mean": 0.016246721148490906, "step": 1254 }, { "clip_ratio/high_max": 0.015558094717562199, "clip_ratio/high_mean": 0.015558094717562199, "clip_ratio/low_mean": 0.015763227711431682, "clip_ratio/low_min": 0.015763227711431682, "clip_ratio/region_mean": 0.03132132242899388, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 95.375, "completions/mean_terminated_length": 95.375, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.12418541312217712, "epoch": 0.05040767964011728, "frac_reward_zero_std": 0.0, "grad_norm": 4.520086765289307, "learning_rate": 6.200000000000001e-06, "loss": -0.0, "num_tokens": 2832143.0, "reward": 0.5924432277679443, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6753093600273132, "reward_meter_std": 0.18275485932826996, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.18791036307811737, "reward_total_composite_mean": 0.5924432277679443, "reward_total_composite_std": 0.18791036307811737, "reward_total_mean": 0.5924432277679443, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6753093600273132, "rewards/meter/std": 0.18275485932826996, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.5924432277679443, "rewards/total_composite/std": 0.18791036307811737, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0073623657226562, "sampling/importance_sampling_ratio/min": 0.12284202128648758, "sampling/sampling_logp_difference/max": 2.096856117248535, "sampling/sampling_logp_difference/mean": 0.036074861884117126, "step": 1255 }, { "clip_ratio/high_max": 0.015424644108861685, "clip_ratio/high_mean": 0.015424644108861685, "clip_ratio/low_mean": 0.024938606191426516, "clip_ratio/low_min": 0.024938606191426516, "clip_ratio/region_mean": 0.0403632503002882, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 64.875, "completions/mean_terminated_length": 64.875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.3142077159136534, "epoch": 0.050447845121902236, "frac_reward_zero_std": 0.0, "grad_norm": 4.7573018074035645, "learning_rate": 6.196969696969698e-06, "loss": 0.016, "num_tokens": 2833894.0, "reward": 0.34324562549591064, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.34324562549591064, "reward_meter_std": 0.38487473130226135, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.38487470149993896, "reward_total_composite_mean": 0.34324562549591064, "reward_total_composite_std": 0.38487473130226135, "reward_total_mean": 0.34324562549591064, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.34324562549591064, "rewards/meter/std": 0.38487473130226135, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.34324562549591064, "rewards/total_composite/std": 0.38487473130226135, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0009076595306396, "sampling/importance_sampling_ratio/min": 0.05577996000647545, "sampling/sampling_logp_difference/max": 2.886340618133545, "sampling/sampling_logp_difference/mean": 0.06118984892964363, "step": 1256 }, { "clip_ratio/high_max": 0.011482007801532745, "clip_ratio/high_mean": 0.011482007801532745, "clip_ratio/low_mean": 0.03709893091581762, "clip_ratio/low_min": 0.03709893091581762, "clip_ratio/region_mean": 0.048580938717350364, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.25, "completions/mean_terminated_length": 33.25, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.2989681512117386, "epoch": 0.05048801060368719, "frac_reward_zero_std": 0.0, "grad_norm": 11.342290878295898, "learning_rate": 6.1939393939393944e-06, "loss": 0.0292, "num_tokens": 2835400.0, "reward": 0.5455847978591919, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5455847978591919, "reward_meter_std": 0.46300008893013, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4630000591278076, "reward_total_composite_mean": 0.5455847978591919, "reward_total_composite_std": 0.46300008893013, "reward_total_mean": 0.5455847978591919, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5455847978591919, "rewards/meter/std": 0.46300008893013, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5455847978591919, "rewards/total_composite/std": 0.46300008893013, "sampling/importance_sampling_ratio/max": 1.4548543691635132, "sampling/importance_sampling_ratio/mean": 1.0081783533096313, "sampling/importance_sampling_ratio/min": 0.3910868465900421, "sampling/sampling_logp_difference/max": 0.9388256072998047, "sampling/sampling_logp_difference/mean": 0.05044575408101082, "step": 1257 }, { "clip_ratio/high_max": 0.0009433962404727936, "clip_ratio/high_mean": 0.0009433962404727936, "clip_ratio/low_mean": 0.006089399073971435, "clip_ratio/low_min": 0.006089399073971435, "clip_ratio/region_mean": 0.007032795314444229, "completions/clipped_ratio": 0.0, "completions/max_length": 270.0, "completions/max_terminated_length": 270.0, "completions/mean_length": 265.25, "completions/mean_terminated_length": 265.25, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.06276637688279152, "epoch": 0.050528176085472144, "frac_reward_zero_std": 0.0, "grad_norm": 1.4305702447891235, "learning_rate": 6.190909090909092e-06, "loss": 0.003, "num_tokens": 2839026.0, "reward": 0.4458819031715393, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6363636255264282, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982133507728577, "reward_meter_std": 0.0003504555206745863, "reward_repeat_penalty_mean": 0.701923131942749, "reward_repeat_penalty_std": 0.027196412906050682, "reward_std": 0.017329316586256027, "reward_total_composite_mean": 0.4458819031715393, "reward_total_composite_std": 0.017329318448901176, "reward_total_mean": 0.4458819031715393, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6363636255264282, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982133507728577, "rewards/meter/std": 0.0003504555206745863, "rewards/repeat_penalty/mean": 0.701923131942749, "rewards/repeat_penalty/std": 0.027196412906050682, "rewards/total_composite/mean": 0.4458819031715393, "rewards/total_composite/std": 0.017329318448901176, "sampling/importance_sampling_ratio/max": 1.5016297101974487, "sampling/importance_sampling_ratio/mean": 1.0004147291183472, "sampling/importance_sampling_ratio/min": 0.2913481593132019, "sampling/sampling_logp_difference/max": 1.233236312866211, "sampling/sampling_logp_difference/mean": 0.011669746600091457, "step": 1258 }, { "clip_ratio/high_max": 0.015308146830648184, "clip_ratio/high_mean": 0.015308146830648184, "clip_ratio/low_mean": 0.01721830479800701, "clip_ratio/low_min": 0.01721830479800701, "clip_ratio/region_mean": 0.032526451628655195, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 107.625, "completions/mean_terminated_length": 107.625, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.18995575420558453, "epoch": 0.0505683415672571, "frac_reward_zero_std": 0.0, "grad_norm": 3.326974630355835, "learning_rate": 6.187878787878788e-06, "loss": 0.013, "num_tokens": 2841255.0, "reward": 0.4992244839668274, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5407921671867371, "reward_meter_std": 0.3177524209022522, "reward_repeat_penalty_mean": 0.925000011920929, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.31581348180770874, "reward_total_composite_mean": 0.4992244839668274, "reward_total_composite_std": 0.31581348180770874, "reward_total_mean": 0.4992244839668274, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5407921671867371, "rewards/meter/std": 0.3177524209022522, "rewards/repeat_penalty/mean": 0.925000011920929, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.4992244839668274, "rewards/total_composite/std": 0.31581348180770874, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0075656175613403, "sampling/importance_sampling_ratio/min": 0.24779127538204193, "sampling/sampling_logp_difference/max": 1.4409921169281006, "sampling/sampling_logp_difference/mean": 0.04232693091034889, "step": 1259 }, { "clip_ratio/high_max": 0.028182906797155738, "clip_ratio/high_mean": 0.028182906797155738, "clip_ratio/low_mean": 0.0029761905316263437, "clip_ratio/low_min": 0.0029761905316263437, "clip_ratio/region_mean": 0.03115909732878208, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 40.875, "completions/mean_terminated_length": 40.875, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.174057693220675, "epoch": 0.05060850704904205, "frac_reward_zero_std": 0.0, "grad_norm": 2.4279935359954834, "learning_rate": 6.184848484848485e-06, "loss": 0.0127, "num_tokens": 2842846.0, "reward": 0.9946430921554565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946430921554565, "reward_meter_std": 0.006536941975355148, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00653693825006485, "reward_total_composite_mean": 0.9946430921554565, "reward_total_composite_std": 0.006536941975355148, "reward_total_mean": 0.9946430921554565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946430921554565, "rewards/meter/std": 0.006536941975355148, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946430921554565, "rewards/total_composite/std": 0.006536941975355148, "sampling/importance_sampling_ratio/max": 1.3470659255981445, "sampling/importance_sampling_ratio/mean": 0.9979214072227478, "sampling/importance_sampling_ratio/min": 0.3126678168773651, "sampling/sampling_logp_difference/max": 1.162613868713379, "sampling/sampling_logp_difference/mean": 0.03152947872877121, "step": 1260 }, { "clip_ratio/high_max": 0.027944298926740885, "clip_ratio/high_mean": 0.027944298926740885, "clip_ratio/low_mean": 0.024214978329837322, "clip_ratio/low_min": 0.024214978329837322, "clip_ratio/region_mean": 0.05215927725657821, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 93.375, "completions/mean_terminated_length": 93.375, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.34710839204490185, "epoch": 0.050648672530827006, "frac_reward_zero_std": 0.0, "grad_norm": 5.053354263305664, "learning_rate": 6.181818181818182e-06, "loss": 0.0029, "num_tokens": 2844913.0, "reward": 0.3705918788909912, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.39358216524124146, "reward_meter_std": 0.3734220564365387, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.34063172340393066, "reward_total_composite_mean": 0.3705918788909912, "reward_total_composite_std": 0.34063175320625305, "reward_total_mean": 0.3705918788909912, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.39358216524124146, "rewards/meter/std": 0.3734220564365387, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.3705918788909912, "rewards/total_composite/std": 0.34063175320625305, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0092734098434448, "sampling/importance_sampling_ratio/min": 0.26026561856269836, "sampling/sampling_logp_difference/max": 1.3460525274276733, "sampling/sampling_logp_difference/mean": 0.052313853055238724, "step": 1261 }, { "clip_ratio/high_max": 0.018799963407218456, "clip_ratio/high_mean": 0.018799963407218456, "clip_ratio/low_mean": 0.010547005338594317, "clip_ratio/low_min": 0.010547005338594317, "clip_ratio/region_mean": 0.029346968745812774, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 158.0, "completions/mean_terminated_length": 158.0, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.0898361736908555, "epoch": 0.05068883801261196, "frac_reward_zero_std": 0.0, "grad_norm": 2.785076856613159, "learning_rate": 6.17878787878788e-06, "loss": 0.0561, "num_tokens": 2847577.0, "reward": 0.32526856660842896, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.504827618598938, "reward_meter_std": 0.3096640408039093, "reward_repeat_penalty_mean": 0.7850378751754761, "reward_repeat_penalty_std": 0.07399878650903702, "reward_std": 0.17344774305820465, "reward_total_composite_mean": 0.32526856660842896, "reward_total_composite_std": 0.17344774305820465, "reward_total_mean": 0.32526856660842896, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.504827618598938, "rewards/meter/std": 0.3096640408039093, "rewards/repeat_penalty/mean": 0.7850378751754761, "rewards/repeat_penalty/std": 0.07399878650903702, "rewards/total_composite/mean": 0.32526856660842896, "rewards/total_composite/std": 0.17344774305820465, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9964918494224548, "sampling/importance_sampling_ratio/min": 0.3024720251560211, "sampling/sampling_logp_difference/max": 1.1957664489746094, "sampling/sampling_logp_difference/mean": 0.028311705216765404, "step": 1262 }, { "clip_ratio/high_max": 0.006790850544348359, "clip_ratio/high_mean": 0.006790850544348359, "clip_ratio/low_mean": 0.007349682127824053, "clip_ratio/low_min": 0.007349682127824053, "clip_ratio/region_mean": 0.014140532672172412, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 272.625, "completions/mean_terminated_length": 272.625, "completions/min_length": 262.0, "completions/min_terminated_length": 262.0, "entropy": 0.11504366714507341, "epoch": 0.050729003494396914, "frac_reward_zero_std": 0.0, "grad_norm": 1.8219730854034424, "learning_rate": 6.175757575757576e-06, "loss": 0.0009, "num_tokens": 2851398.0, "reward": 0.4640815854072571, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6363636255264282, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9851720929145813, "reward_meter_std": 0.025058675557374954, "reward_repeat_penalty_mean": 0.7403846383094788, "reward_repeat_penalty_std": 0.07047118246555328, "reward_std": 0.044502876698970795, "reward_total_composite_mean": 0.4640815854072571, "reward_total_composite_std": 0.0445028655230999, "reward_total_mean": 0.4640815854072571, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6363636255264282, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9851720929145813, "rewards/meter/std": 0.025058675557374954, "rewards/repeat_penalty/mean": 0.7403846383094788, "rewards/repeat_penalty/std": 0.07047118246555328, "rewards/total_composite/mean": 0.4640815854072571, "rewards/total_composite/std": 0.0445028655230999, "sampling/importance_sampling_ratio/max": 1.9002686738967896, "sampling/importance_sampling_ratio/mean": 0.9996957182884216, "sampling/importance_sampling_ratio/min": 0.1570604145526886, "sampling/sampling_logp_difference/max": 1.8511247634887695, "sampling/sampling_logp_difference/mean": 0.023550348356366158, "step": 1263 }, { "clip_ratio/high_max": 0.00657894741743803, "clip_ratio/high_mean": 0.00657894741743803, "clip_ratio/low_mean": 0.021268213167786598, "clip_ratio/low_min": 0.021268213167786598, "clip_ratio/region_mean": 0.02784716058522463, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 58.5, "completions/mean_terminated_length": 58.5, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.23071717843413353, "epoch": 0.05076916897618187, "frac_reward_zero_std": 0.0, "grad_norm": 6.026924133300781, "learning_rate": 6.1727272727272735e-06, "loss": 0.0151, "num_tokens": 2853090.0, "reward": 0.2589147090911865, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2589147090911865, "reward_meter_std": 0.21604736149311066, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.21604733169078827, "reward_total_composite_mean": 0.2589147090911865, "reward_total_composite_std": 0.21604736149311066, "reward_total_mean": 0.2589147090911865, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2589147090911865, "rewards/meter/std": 0.21604736149311066, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.2589147090911865, "rewards/total_composite/std": 0.21604736149311066, "sampling/importance_sampling_ratio/max": 1.932633399963379, "sampling/importance_sampling_ratio/mean": 1.0070089101791382, "sampling/importance_sampling_ratio/min": 0.4005609452724457, "sampling/sampling_logp_difference/max": 0.9148893356323242, "sampling/sampling_logp_difference/mean": 0.039983682334423065, "step": 1264 }, { "clip_ratio/high_max": 0.05259183724410832, "clip_ratio/high_mean": 0.05259183724410832, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/region_mean": 0.05597021570429206, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 36.0, "completions/mean_terminated_length": 36.0, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.4582943990826607, "epoch": 0.05080933445796682, "frac_reward_zero_std": 0.0, "grad_norm": 16.352861404418945, "learning_rate": 6.16969696969697e-06, "loss": 0.0293, "num_tokens": 2854666.0, "reward": 0.8565147519111633, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8565147519111633, "reward_meter_std": 0.3440394103527069, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3440393805503845, "reward_total_composite_mean": 0.8565147519111633, "reward_total_composite_std": 0.3440394103527069, "reward_total_mean": 0.8565147519111633, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8565147519111633, "rewards/meter/std": 0.3440394103527069, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8565147519111633, "rewards/total_composite/std": 0.3440394103527069, "sampling/importance_sampling_ratio/max": 1.7478100061416626, "sampling/importance_sampling_ratio/mean": 0.9966188073158264, "sampling/importance_sampling_ratio/min": 0.327095627784729, "sampling/sampling_logp_difference/max": 1.1175026893615723, "sampling/sampling_logp_difference/mean": 0.057592298835515976, "step": 1265 }, { "clip_ratio/high_max": 0.016067266347818077, "clip_ratio/high_mean": 0.016067266347818077, "clip_ratio/low_mean": 0.025986650260165334, "clip_ratio/low_min": 0.025986650260165334, "clip_ratio/region_mean": 0.04205391660798341, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 77.375, "completions/mean_terminated_length": 77.375, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.2823401764035225, "epoch": 0.050849499939751776, "frac_reward_zero_std": 0.0, "grad_norm": 3.7630727291107178, "learning_rate": 6.166666666666667e-06, "loss": 0.0058, "num_tokens": 2856589.0, "reward": 0.9948470592498779, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948470592498779, "reward_meter_std": 0.0029209780041128397, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0029209787026047707, "reward_total_composite_mean": 0.9948470592498779, "reward_total_composite_std": 0.0029209780041128397, "reward_total_mean": 0.9948470592498779, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948470592498779, "rewards/meter/std": 0.0029209780041128397, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948470592498779, "rewards/total_composite/std": 0.0029209780041128397, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0063776969909668, "sampling/importance_sampling_ratio/min": 0.28769612312316895, "sampling/sampling_logp_difference/max": 1.2458505630493164, "sampling/sampling_logp_difference/mean": 0.04110397398471832, "step": 1266 }, { "clip_ratio/high_max": 0.035607312340289354, "clip_ratio/high_mean": 0.035607312340289354, "clip_ratio/low_mean": 0.02490918291732669, "clip_ratio/low_min": 0.02490918291732669, "clip_ratio/region_mean": 0.06051649525761604, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 63.875, "completions/mean_terminated_length": 63.875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.5581622421741486, "epoch": 0.05088966542153673, "frac_reward_zero_std": 0.0, "grad_norm": 7.6448774337768555, "learning_rate": 6.163636363636364e-06, "loss": 0.0264, "num_tokens": 2858420.0, "reward": 0.6281806230545044, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6281806230545044, "reward_meter_std": 0.4045161306858063, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4045161306858063, "reward_total_composite_mean": 0.6281806230545044, "reward_total_composite_std": 0.4045161306858063, "reward_total_mean": 0.6281806230545044, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6281806230545044, "rewards/meter/std": 0.4045161306858063, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6281806230545044, "rewards/total_composite/std": 0.4045161306858063, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0073941946029663, "sampling/importance_sampling_ratio/min": 0.21431100368499756, "sampling/sampling_logp_difference/max": 1.5403270721435547, "sampling/sampling_logp_difference/mean": 0.08960650116205215, "step": 1267 }, { "clip_ratio/high_max": 0.028985073906369507, "clip_ratio/high_mean": 0.028985073906369507, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/region_mean": 0.032409731415100396, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 76.25, "completions/mean_terminated_length": 76.25, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.1600959748029709, "epoch": 0.050929830903321684, "frac_reward_zero_std": 0.0, "grad_norm": 6.307591438293457, "learning_rate": 6.160606060606062e-06, "loss": -0.008, "num_tokens": 2860190.0, "reward": 0.9219943881034851, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9219943881034851, "reward_meter_std": 0.2038099318742752, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2038099467754364, "reward_total_composite_mean": 0.9219943881034851, "reward_total_composite_std": 0.2038099318742752, "reward_total_mean": 0.9219943881034851, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9219943881034851, "rewards/meter/std": 0.2038099318742752, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9219943881034851, "rewards/total_composite/std": 0.2038099318742752, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.99973064661026, "sampling/importance_sampling_ratio/min": 0.3323935270309448, "sampling/sampling_logp_difference/max": 1.101435661315918, "sampling/sampling_logp_difference/mean": 0.03656025603413582, "step": 1268 }, { "clip_ratio/high_max": 0.034388539381325245, "clip_ratio/high_mean": 0.034388539381325245, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/region_mean": 0.037766917841508985, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.09085107641294599, "epoch": 0.05096999638510664, "frac_reward_zero_std": 0.0, "grad_norm": 16.08408546447754, "learning_rate": 6.157575757575758e-06, "loss": 0.0401, "num_tokens": 2862222.0, "reward": 0.8981978893280029, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8981978893280029, "reward_meter_std": 0.22402629256248474, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22402627766132355, "reward_total_composite_mean": 0.8981978893280029, "reward_total_composite_std": 0.22402629256248474, "reward_total_mean": 0.8981978893280029, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8981978893280029, "rewards/meter/std": 0.22402629256248474, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8981978893280029, "rewards/total_composite/std": 0.22402629256248474, "sampling/importance_sampling_ratio/max": 1.8017491102218628, "sampling/importance_sampling_ratio/mean": 0.9943932890892029, "sampling/importance_sampling_ratio/min": 0.0958891436457634, "sampling/sampling_logp_difference/max": 2.344562530517578, "sampling/sampling_logp_difference/mean": 0.03928679972887039, "step": 1269 }, { "clip_ratio/high_max": 0.006335443118587136, "clip_ratio/high_mean": 0.006335443118587136, "clip_ratio/low_mean": 0.004985210340237245, "clip_ratio/low_min": 0.004985210340237245, "clip_ratio/region_mean": 0.011320653458824381, "completions/clipped_ratio": 0.0, "completions/max_length": 388.0, "completions/max_terminated_length": 388.0, "completions/mean_length": 375.25, "completions/mean_terminated_length": 375.25, "completions/min_length": 359.0, "completions/min_terminated_length": 359.0, "entropy": 0.07593974284827709, "epoch": 0.05101016186689159, "frac_reward_zero_std": 0.0, "grad_norm": 1.1973329782485962, "learning_rate": 6.154545454545455e-06, "loss": -0.0005, "num_tokens": 2866664.0, "reward": 0.35947954654693604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5263158082962036, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982415437698364, "reward_meter_std": 0.00024763745022937655, "reward_repeat_penalty_mean": 0.6842105388641357, "reward_repeat_penalty_std": 0.07443230599164963, "reward_std": 0.039129648357629776, "reward_total_composite_mean": 0.35947954654693604, "reward_total_composite_std": 0.039129648357629776, "reward_total_mean": 0.35947954654693604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5263158082962036, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982415437698364, "rewards/meter/std": 0.00024763745022937655, "rewards/repeat_penalty/mean": 0.6842105388641357, "rewards/repeat_penalty/std": 0.07443230599164963, "rewards/total_composite/mean": 0.35947954654693604, "rewards/total_composite/std": 0.039129648357629776, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016210079193115, "sampling/importance_sampling_ratio/min": 0.14204660058021545, "sampling/sampling_logp_difference/max": 1.9516000747680664, "sampling/sampling_logp_difference/mean": 0.01408437080681324, "step": 1270 }, { "clip_ratio/high_max": 0.00831140368245542, "clip_ratio/high_mean": 0.00831140368245542, "clip_ratio/low_mean": 0.0032056551426649094, "clip_ratio/low_min": 0.0032056551426649094, "clip_ratio/region_mean": 0.01151705882512033, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.125, "completions/mean_terminated_length": 76.125, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.13653591368347406, "epoch": 0.051050327348676545, "frac_reward_zero_std": 0.0, "grad_norm": 3.943135976791382, "learning_rate": 6.151515151515152e-06, "loss": 0.0134, "num_tokens": 2868449.0, "reward": 0.9946237802505493, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946237802505493, "reward_meter_std": 0.0009093984263017774, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009094090783037245, "reward_total_composite_mean": 0.9946237802505493, "reward_total_composite_std": 0.0009093984263017774, "reward_total_mean": 0.9946237802505493, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946237802505493, "rewards/meter/std": 0.0009093984263017774, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946237802505493, "rewards/total_composite/std": 0.0009093984263017774, "sampling/importance_sampling_ratio/max": 1.8516945838928223, "sampling/importance_sampling_ratio/mean": 1.0047322511672974, "sampling/importance_sampling_ratio/min": 0.4211141765117645, "sampling/sampling_logp_difference/max": 0.8648512959480286, "sampling/sampling_logp_difference/mean": 0.02133559249341488, "step": 1271 }, { "clip_ratio/high_max": 0.036419710610061884, "clip_ratio/high_mean": 0.036419710610061884, "clip_ratio/low_mean": 0.010900080436840653, "clip_ratio/low_min": 0.010900080436840653, "clip_ratio/region_mean": 0.04731979104690254, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.26362900249660015, "epoch": 0.0510904928304615, "frac_reward_zero_std": 0.0, "grad_norm": 5.4271039962768555, "learning_rate": 6.148484848484849e-06, "loss": 0.0295, "num_tokens": 2870337.0, "reward": 0.7235432863235474, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7235432863235474, "reward_meter_std": 0.3634967803955078, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3634967505931854, "reward_total_composite_mean": 0.7235432863235474, "reward_total_composite_std": 0.3634967803955078, "reward_total_mean": 0.7235432863235474, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7235432863235474, "rewards/meter/std": 0.3634967803955078, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7235432863235474, "rewards/total_composite/std": 0.3634967803955078, "sampling/importance_sampling_ratio/max": 1.7601451873779297, "sampling/importance_sampling_ratio/mean": 0.9991426467895508, "sampling/importance_sampling_ratio/min": 0.024505402892827988, "sampling/sampling_logp_difference/max": 3.7088615894317627, "sampling/sampling_logp_difference/mean": 0.06481360644102097, "step": 1272 }, { "clip_ratio/high_max": 0.00823544233571738, "clip_ratio/high_mean": 0.00823544233571738, "clip_ratio/low_mean": 0.004043337190523744, "clip_ratio/low_min": 0.004043337190523744, "clip_ratio/region_mean": 0.012278779526241124, "completions/clipped_ratio": 0.0, "completions/max_length": 157.0, "completions/max_terminated_length": 157.0, "completions/mean_length": 152.875, "completions/mean_terminated_length": 152.875, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.10434958059340715, "epoch": 0.05113065831224645, "frac_reward_zero_std": 0.0, "grad_norm": 3.146674394607544, "learning_rate": 6.1454545454545454e-06, "loss": 0.0059, "num_tokens": 2872992.0, "reward": 0.855392336845398, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979577660560608, "reward_meter_std": 0.0006660494254902005, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005708927637897432, "reward_total_composite_mean": 0.855392336845398, "reward_total_composite_std": 0.0005708994576707482, "reward_total_mean": 0.855392336845398, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979577660560608, "rewards/meter/std": 0.0006660494254902005, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.855392336845398, "rewards/total_composite/std": 0.0005708994576707482, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9994745850563049, "sampling/importance_sampling_ratio/min": 0.25055789947509766, "sampling/sampling_logp_difference/max": 1.3840652704238892, "sampling/sampling_logp_difference/mean": 0.01988840475678444, "step": 1273 }, { "clip_ratio/high_max": 0.008859357796609402, "clip_ratio/high_mean": 0.008859357796609402, "clip_ratio/low_mean": 0.0029761905316263437, "clip_ratio/low_min": 0.0029761905316263437, "clip_ratio/region_mean": 0.011835548328235745, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 42.125, "completions/mean_terminated_length": 42.125, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.1509600691497326, "epoch": 0.05117082379403141, "frac_reward_zero_std": 0.0, "grad_norm": 9.43035888671875, "learning_rate": 6.142424242424243e-06, "loss": -0.0028, "num_tokens": 2874585.0, "reward": 0.9962718486785889, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962718486785889, "reward_meter_std": 0.0009969644015654922, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009969675447791815, "reward_total_composite_mean": 0.9962718486785889, "reward_total_composite_std": 0.0009969644015654922, "reward_total_mean": 0.9962718486785889, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962718486785889, "rewards/meter/std": 0.0009969644015654922, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9962718486785889, "rewards/total_composite/std": 0.0009969644015654922, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0059658288955688, "sampling/importance_sampling_ratio/min": 0.547152042388916, "sampling/sampling_logp_difference/max": 0.7795038223266602, "sampling/sampling_logp_difference/mean": 0.02699565887451172, "step": 1274 }, { "clip_ratio/high_max": 0.012963371758814901, "clip_ratio/high_mean": 0.012963371758814901, "clip_ratio/low_mean": 0.007767904899083078, "clip_ratio/low_min": 0.007767904899083078, "clip_ratio/region_mean": 0.02073127665789798, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 128.125, "completions/mean_terminated_length": 128.125, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.20628871954977512, "epoch": 0.05121098927581636, "frac_reward_zero_std": 0.0, "grad_norm": 3.5774126052856445, "learning_rate": 6.139393939393939e-06, "loss": -0.0071, "num_tokens": 2877026.0, "reward": 0.8314633369445801, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9509094953536987, "reward_meter_std": 0.036813125014305115, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.04581277072429657, "reward_total_composite_mean": 0.8314633369445801, "reward_total_composite_std": 0.04581277072429657, "reward_total_mean": 0.8314633369445801, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9509094953536987, "rewards/meter/std": 0.036813125014305115, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8314633369445801, "rewards/total_composite/std": 0.04581277072429657, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0062079429626465, "sampling/importance_sampling_ratio/min": 0.32652077078819275, "sampling/sampling_logp_difference/max": 1.1192617416381836, "sampling/sampling_logp_difference/mean": 0.035852231085300446, "step": 1275 }, { "clip_ratio/high_max": 0.028633129317313433, "clip_ratio/high_mean": 0.028633129317313433, "clip_ratio/low_mean": 0.0324383438564837, "clip_ratio/low_min": 0.0324383438564837, "clip_ratio/region_mean": 0.06107147317379713, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 89.25, "completions/mean_terminated_length": 89.25, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.8716921396553516, "epoch": 0.051251154757601315, "frac_reward_zero_std": 0.0, "grad_norm": 7.507855415344238, "learning_rate": 6.136363636363637e-06, "loss": 0.0392, "num_tokens": 2879100.0, "reward": 0.3967258036136627, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.414013534784317, "reward_meter_std": 0.36701732873916626, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.3643990457057953, "reward_total_composite_mean": 0.3967258036136627, "reward_total_composite_std": 0.3643990457057953, "reward_total_mean": 0.3967258036136627, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.414013534784317, "rewards/meter/std": 0.36701732873916626, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.3967258036136627, "rewards/total_composite/std": 0.3643990457057953, "sampling/importance_sampling_ratio/max": 1.9257968664169312, "sampling/importance_sampling_ratio/mean": 1.0259032249450684, "sampling/importance_sampling_ratio/min": 0.19726145267486572, "sampling/sampling_logp_difference/max": 1.623225212097168, "sampling/sampling_logp_difference/mean": 0.09024675190448761, "step": 1276 }, { "clip_ratio/high_max": 0.02701578661799431, "clip_ratio/high_mean": 0.02701578661799431, "clip_ratio/low_mean": 0.012976306956261396, "clip_ratio/low_min": 0.012976306956261396, "clip_ratio/region_mean": 0.039992093574255705, "completions/clipped_ratio": 0.0, "completions/max_length": 184.0, "completions/max_terminated_length": 184.0, "completions/mean_length": 177.75, "completions/mean_terminated_length": 177.75, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.3461093604564667, "epoch": 0.05129132023938627, "frac_reward_zero_std": 0.0, "grad_norm": 5.259714126586914, "learning_rate": 6.133333333333334e-06, "loss": 0.0264, "num_tokens": 2881970.0, "reward": 0.6403263807296753, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8953356742858887, "reward_meter_std": 0.2748320698738098, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.05143444612622261, "reward_std": 0.19905687868595123, "reward_total_composite_mean": 0.6403263807296753, "reward_total_composite_std": 0.19905686378479004, "reward_total_mean": 0.6403263807296753, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8953356742858887, "rewards/meter/std": 0.2748320698738098, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.6403263807296753, "rewards/total_composite/std": 0.19905686378479004, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0098189115524292, "sampling/importance_sampling_ratio/min": 0.2969248294830322, "sampling/sampling_logp_difference/max": 1.2142763137817383, "sampling/sampling_logp_difference/mean": 0.04297341778874397, "step": 1277 }, { "clip_ratio/high_max": 0.054509096313267946, "clip_ratio/high_mean": 0.054509096313267946, "clip_ratio/low_mean": 0.012557290028780699, "clip_ratio/low_min": 0.012557290028780699, "clip_ratio/region_mean": 0.06706638634204865, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 48.75, "completions/mean_terminated_length": 48.75, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.2942120973020792, "epoch": 0.05133148572117122, "frac_reward_zero_std": 0.0, "grad_norm": 10.583206176757812, "learning_rate": 6.130303030303031e-06, "loss": 0.0461, "num_tokens": 2883696.0, "reward": 0.8016507625579834, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8016507625579834, "reward_meter_std": 0.21309679746627808, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.21309679746627808, "reward_total_composite_mean": 0.8016507625579834, "reward_total_composite_std": 0.21309679746627808, "reward_total_mean": 0.8016507625579834, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8016507625579834, "rewards/meter/std": 0.21309679746627808, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8016507625579834, "rewards/total_composite/std": 0.21309679746627808, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.992801308631897, "sampling/importance_sampling_ratio/min": 0.15800006687641144, "sampling/sampling_logp_difference/max": 1.8451597690582275, "sampling/sampling_logp_difference/mean": 0.07192375510931015, "step": 1278 }, { "clip_ratio/high_max": 0.0231478811474517, "clip_ratio/high_mean": 0.0231478811474517, "clip_ratio/low_mean": 0.010369318537414074, "clip_ratio/low_min": 0.010369318537414074, "clip_ratio/region_mean": 0.03351719968486577, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 86.875, "completions/mean_terminated_length": 86.875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.20182008855044842, "epoch": 0.05137165120295618, "frac_reward_zero_std": 0.0, "grad_norm": 4.000077247619629, "learning_rate": 6.127272727272727e-06, "loss": -0.028, "num_tokens": 2885855.0, "reward": 0.625595211982727, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7521673440933228, "reward_meter_std": 0.3259941041469574, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.1511857807636261, "reward_std": 0.32554712891578674, "reward_total_composite_mean": 0.625595211982727, "reward_total_composite_std": 0.32554712891578674, "reward_total_mean": 0.625595211982727, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7521673440933228, "rewards/meter/std": 0.3259941041469574, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.1511857807636261, "rewards/total_composite/mean": 0.625595211982727, "rewards/total_composite/std": 0.32554712891578674, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005535364151001, "sampling/importance_sampling_ratio/min": 0.45753392577171326, "sampling/sampling_logp_difference/max": 1.77433443069458, "sampling/sampling_logp_difference/mean": 0.03764592856168747, "step": 1279 }, { "clip_ratio/high_max": 0.020130750257521868, "clip_ratio/high_mean": 0.020130750257521868, "clip_ratio/low_mean": 0.035150347743183374, "clip_ratio/low_min": 0.035150347743183374, "clip_ratio/region_mean": 0.05528109800070524, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 110.625, "completions/mean_terminated_length": 110.625, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.3459767811000347, "epoch": 0.05141181668474113, "frac_reward_zero_std": 0.0, "grad_norm": 5.375721454620361, "learning_rate": 6.1242424242424245e-06, "loss": 0.0056, "num_tokens": 2888100.0, "reward": 0.3231015205383301, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.32401683926582336, "reward_meter_std": 0.4113844931125641, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.41212278604507446, "reward_total_composite_mean": 0.3231015205383301, "reward_total_composite_std": 0.41212281584739685, "reward_total_mean": 0.3231015205383301, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.32401683926582336, "rewards/meter/std": 0.4113844931125641, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.3231015205383301, "rewards/total_composite/std": 0.41212281584739685, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015076398849487, "sampling/importance_sampling_ratio/min": 0.012362679466605186, "sampling/sampling_logp_difference/max": 4.393073081970215, "sampling/sampling_logp_difference/mean": 0.07019690424203873, "step": 1280 }, { "clip_ratio/high_max": 0.03385740250814706, "clip_ratio/high_mean": 0.03385740250814706, "clip_ratio/low_mean": 0.005307539715431631, "clip_ratio/low_min": 0.005307539715431631, "clip_ratio/region_mean": 0.03916494222357869, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 70.375, "completions/mean_terminated_length": 70.375, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2260036300867796, "epoch": 0.051451982166526085, "frac_reward_zero_std": 0.0, "grad_norm": 7.6968994140625, "learning_rate": 6.121212121212121e-06, "loss": 0.0157, "num_tokens": 2889991.0, "reward": 0.9908488988876343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9908488988876343, "reward_meter_std": 0.0105263227596879, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010526325553655624, "reward_total_composite_mean": 0.9908488988876343, "reward_total_composite_std": 0.0105263227596879, "reward_total_mean": 0.9908488988876343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9908488988876343, "rewards/meter/std": 0.0105263227596879, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9908488988876343, "rewards/total_composite/std": 0.0105263227596879, "sampling/importance_sampling_ratio/max": 1.7191059589385986, "sampling/importance_sampling_ratio/mean": 0.9948356747627258, "sampling/importance_sampling_ratio/min": 0.19431231915950775, "sampling/sampling_logp_difference/max": 1.6382884979248047, "sampling/sampling_logp_difference/mean": 0.044087208807468414, "step": 1281 }, { "clip_ratio/high_max": 0.013975600013509393, "clip_ratio/high_mean": 0.013975600013509393, "clip_ratio/low_mean": 0.010263397532980889, "clip_ratio/low_min": 0.010263397532980889, "clip_ratio/region_mean": 0.024238997546490282, "completions/clipped_ratio": 0.0, "completions/max_length": 311.0, "completions/max_terminated_length": 311.0, "completions/mean_length": 303.875, "completions/mean_terminated_length": 303.875, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "entropy": 0.20566194783896208, "epoch": 0.05149214764831104, "frac_reward_zero_std": 0.0, "grad_norm": 2.044459581375122, "learning_rate": 6.118181818181819e-06, "loss": 0.0068, "num_tokens": 2894526.0, "reward": 0.43306678533554077, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5714285969734192, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9781317114830017, "reward_meter_std": 0.027103744447231293, "reward_repeat_penalty_mean": 0.7749999761581421, "reward_repeat_penalty_std": 0.11233452707529068, "reward_std": 0.0634385421872139, "reward_total_composite_mean": 0.43306678533554077, "reward_total_composite_std": 0.0634385421872139, "reward_total_mean": 0.43306678533554077, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5714285969734192, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9781317114830017, "rewards/meter/std": 0.027103744447231293, "rewards/repeat_penalty/mean": 0.7749999761581421, "rewards/repeat_penalty/std": 0.11233452707529068, "rewards/total_composite/mean": 0.43306678533554077, "rewards/total_composite/std": 0.0634385421872139, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0035651922225952, "sampling/importance_sampling_ratio/min": 0.0384955108165741, "sampling/sampling_logp_difference/max": 3.257213592529297, "sampling/sampling_logp_difference/mean": 0.03700267896056175, "step": 1282 }, { "clip_ratio/high_max": 0.030571716954000294, "clip_ratio/high_mean": 0.030571716954000294, "clip_ratio/low_mean": 0.009698634734377265, "clip_ratio/low_min": 0.009698634734377265, "clip_ratio/region_mean": 0.04027035168837756, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.625, "completions/mean_terminated_length": 76.625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.43742145597934723, "epoch": 0.05153231313009599, "frac_reward_zero_std": 0.0, "grad_norm": 5.159550666809082, "learning_rate": 6.115151515151516e-06, "loss": 0.0097, "num_tokens": 2896475.0, "reward": 0.9953988790512085, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953988790512085, "reward_meter_std": 0.0025027033407241106, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0025027082301676273, "reward_total_composite_mean": 0.9953988790512085, "reward_total_composite_std": 0.0025027033407241106, "reward_total_mean": 0.9953988790512085, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953988790512085, "rewards/meter/std": 0.0025027033407241106, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953988790512085, "rewards/total_composite/std": 0.0025027033407241106, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0098578929901123, "sampling/importance_sampling_ratio/min": 0.21837389469146729, "sampling/sampling_logp_difference/max": 1.5215466022491455, "sampling/sampling_logp_difference/mean": 0.053737834095954895, "step": 1283 }, { "clip_ratio/high_max": 0.020760516403242946, "clip_ratio/high_mean": 0.020760516403242946, "clip_ratio/low_mean": 0.03277311008423567, "clip_ratio/low_min": 0.03277311008423567, "clip_ratio/region_mean": 0.053533626487478614, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 36.0, "completions/mean_terminated_length": 36.0, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.31284985691308975, "epoch": 0.05157247861188095, "frac_reward_zero_std": 0.0, "grad_norm": 8.960368156433105, "learning_rate": 6.112121212121213e-06, "loss": -0.0204, "num_tokens": 2897859.0, "reward": 0.7910915613174438, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7910915613174438, "reward_meter_std": 0.36750105023384094, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.36750105023384094, "reward_total_composite_mean": 0.7910915613174438, "reward_total_composite_std": 0.36750105023384094, "reward_total_mean": 0.7910915613174438, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7910915613174438, "rewards/meter/std": 0.36750105023384094, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7910915613174438, "rewards/total_composite/std": 0.36750105023384094, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.996502161026001, "sampling/importance_sampling_ratio/min": 0.24718981981277466, "sampling/sampling_logp_difference/max": 1.3975987434387207, "sampling/sampling_logp_difference/mean": 0.06175791099667549, "step": 1284 }, { "clip_ratio/high_max": 0.01694088790100068, "clip_ratio/high_mean": 0.01694088790100068, "clip_ratio/low_mean": 0.013498181011527777, "clip_ratio/low_min": 0.013498181011527777, "clip_ratio/region_mean": 0.030439068912528455, "completions/clipped_ratio": 0.0, "completions/max_length": 146.0, "completions/max_terminated_length": 146.0, "completions/mean_length": 133.25, "completions/mean_terminated_length": 133.25, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.272802772000432, "epoch": 0.0516126440936659, "frac_reward_zero_std": 0.0, "grad_norm": 4.468361854553223, "learning_rate": 6.10909090909091e-06, "loss": -0.0018, "num_tokens": 2900285.0, "reward": 0.7135751247406006, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7679275274276733, "reward_meter_std": 0.3219233751296997, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.31732064485549927, "reward_total_composite_mean": 0.7135751247406006, "reward_total_composite_std": 0.3173206150531769, "reward_total_mean": 0.7135751247406006, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7679275274276733, "rewards/meter/std": 0.3219233751296997, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.7135751247406006, "rewards/total_composite/std": 0.3173206150531769, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006422996520996, "sampling/importance_sampling_ratio/min": 0.27716556191444397, "sampling/sampling_logp_difference/max": 1.2831401824951172, "sampling/sampling_logp_difference/mean": 0.04117295891046524, "step": 1285 }, { "clip_ratio/high_max": 0.027195949805900455, "clip_ratio/high_mean": 0.027195949805900455, "clip_ratio/low_mean": 0.007508100010454655, "clip_ratio/low_min": 0.007508100010454655, "clip_ratio/region_mean": 0.03470404981635511, "completions/clipped_ratio": 0.0, "completions/max_length": 233.0, "completions/max_terminated_length": 233.0, "completions/mean_length": 219.875, "completions/mean_terminated_length": 219.875, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "entropy": 0.3109611961990595, "epoch": 0.051652809575450855, "frac_reward_zero_std": 0.0, "grad_norm": 2.972513437271118, "learning_rate": 6.106060606060606e-06, "loss": 0.0163, "num_tokens": 2903604.0, "reward": 0.6425139904022217, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8396387100219727, "reward_meter_std": 0.2860766649246216, "reward_repeat_penalty_mean": 0.8863636255264282, "reward_repeat_penalty_std": 0.06428244709968567, "reward_std": 0.22964133322238922, "reward_total_composite_mean": 0.6425139904022217, "reward_total_composite_std": 0.22964133322238922, "reward_total_mean": 0.6425139904022217, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8396387100219727, "rewards/meter/std": 0.2860766649246216, "rewards/repeat_penalty/mean": 0.8863636255264282, "rewards/repeat_penalty/std": 0.06428244709968567, "rewards/total_composite/mean": 0.6425139904022217, "rewards/total_composite/std": 0.22964133322238922, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0046546459197998, "sampling/importance_sampling_ratio/min": 0.26713767647743225, "sampling/sampling_logp_difference/max": 1.4966444969177246, "sampling/sampling_logp_difference/mean": 0.0455651730298996, "step": 1286 }, { "clip_ratio/high_max": 0.020334920845925808, "clip_ratio/high_mean": 0.020334920845925808, "clip_ratio/low_mean": 0.007304175465833396, "clip_ratio/low_min": 0.007304175465833396, "clip_ratio/region_mean": 0.027639096311759204, "completions/clipped_ratio": 0.0, "completions/max_length": 199.0, "completions/max_terminated_length": 199.0, "completions/mean_length": 189.75, "completions/mean_terminated_length": 189.75, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "entropy": 0.22143690288066864, "epoch": 0.05169297505723581, "frac_reward_zero_std": 0.0, "grad_norm": 3.1675779819488525, "learning_rate": 6.103030303030304e-06, "loss": -0.0036, "num_tokens": 2906858.0, "reward": 0.7119789123535156, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9928793907165527, "reward_meter_std": 0.01269504800438881, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.09848947077989578, "reward_std": 0.0765279084444046, "reward_total_composite_mean": 0.7119789123535156, "reward_total_composite_std": 0.07652789354324341, "reward_total_mean": 0.7119789123535156, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9928793907165527, "rewards/meter/std": 0.01269504800438881, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.09848947077989578, "rewards/total_composite/mean": 0.7119789123535156, "rewards/total_composite/std": 0.07652789354324341, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0034816265106201, "sampling/importance_sampling_ratio/min": 0.04349659010767937, "sampling/sampling_logp_difference/max": 3.135072708129883, "sampling/sampling_logp_difference/mean": 0.0419112965464592, "step": 1287 }, { "clip_ratio/high_max": 0.00875260157044977, "clip_ratio/high_mean": 0.00875260157044977, "clip_ratio/low_mean": 0.008273915416793898, "clip_ratio/low_min": 0.008273915416793898, "clip_ratio/region_mean": 0.017026516987243667, "completions/clipped_ratio": 0.0, "completions/max_length": 275.0, "completions/max_terminated_length": 275.0, "completions/mean_length": 272.0, "completions/mean_terminated_length": 272.0, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.139354033395648, "epoch": 0.05173314053902076, "frac_reward_zero_std": 0.0, "grad_norm": 1.9611791372299194, "learning_rate": 6.1e-06, "loss": 0.0056, "num_tokens": 2910730.0, "reward": 0.6442359685897827, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7777777910232544, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9904348254203796, "reward_meter_std": 0.015252761542797089, "reward_repeat_penalty_mean": 0.8365384340286255, "reward_repeat_penalty_std": 0.08661473542451859, "reward_std": 0.06582249701023102, "reward_total_composite_mean": 0.6442359685897827, "reward_total_composite_std": 0.06582249701023102, "reward_total_mean": 0.6442359685897827, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7777777910232544, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9904348254203796, "rewards/meter/std": 0.015252761542797089, "rewards/repeat_penalty/mean": 0.8365384340286255, "rewards/repeat_penalty/std": 0.08661473542451859, "rewards/total_composite/mean": 0.6442359685897827, "rewards/total_composite/std": 0.06582249701023102, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0058023929595947, "sampling/importance_sampling_ratio/min": 0.10430250316858292, "sampling/sampling_logp_difference/max": 2.2604598999023438, "sampling/sampling_logp_difference/mean": 0.025431334972381592, "step": 1288 }, { "clip_ratio/high_max": 0.0282916008727625, "clip_ratio/high_mean": 0.0282916008727625, "clip_ratio/low_mean": 0.01717152213677764, "clip_ratio/low_min": 0.01717152213677764, "clip_ratio/region_mean": 0.04546312300954014, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.32122630439698696, "epoch": 0.051773306020805716, "frac_reward_zero_std": 0.0, "grad_norm": 5.8379316329956055, "learning_rate": 6.096969696969698e-06, "loss": 0.0464, "num_tokens": 2912411.0, "reward": 0.8378640413284302, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8378640413284302, "reward_meter_std": 0.22375915944576263, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22375912964344025, "reward_total_composite_mean": 0.8378640413284302, "reward_total_composite_std": 0.22375915944576263, "reward_total_mean": 0.8378640413284302, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8378640413284302, "rewards/meter/std": 0.22375915944576263, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8378640413284302, "rewards/total_composite/std": 0.22375915944576263, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0112100839614868, "sampling/importance_sampling_ratio/min": 0.360772967338562, "sampling/sampling_logp_difference/max": 1.0195064544677734, "sampling/sampling_logp_difference/mean": 0.051838938146829605, "step": 1289 }, { "clip_ratio/high_max": 0.01394439465366304, "clip_ratio/high_mean": 0.01394439465366304, "clip_ratio/low_mean": 0.01538592274300754, "clip_ratio/low_min": 0.01538592274300754, "clip_ratio/region_mean": 0.02933031739667058, "completions/clipped_ratio": 0.0, "completions/max_length": 264.0, "completions/max_terminated_length": 264.0, "completions/mean_length": 251.75, "completions/mean_terminated_length": 251.75, "completions/min_length": 245.0, "completions/min_terminated_length": 245.0, "entropy": 0.28565115854144096, "epoch": 0.05181347150259067, "frac_reward_zero_std": 0.0, "grad_norm": 2.28285813331604, "learning_rate": 6.0939393939393946e-06, "loss": 0.0012, "num_tokens": 2915945.0, "reward": 0.012334112077951431, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.017678312957286835, "reward_meter_std": 0.021944953128695488, "reward_repeat_penalty_mean": 0.8269230723381042, "reward_repeat_penalty_std": 0.09859537333250046, "reward_std": 0.01488783210515976, "reward_total_composite_mean": 0.012334112077951431, "reward_total_composite_std": 0.01488783210515976, "reward_total_mean": 0.012334112077951431, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.017678312957286835, "rewards/meter/std": 0.021944953128695488, "rewards/repeat_penalty/mean": 0.8269230723381042, "rewards/repeat_penalty/std": 0.09859537333250046, "rewards/total_composite/mean": 0.012334112077951431, "rewards/total_composite/std": 0.01488783210515976, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0024718046188354, "sampling/importance_sampling_ratio/min": 0.2436685711145401, "sampling/sampling_logp_difference/max": 1.4119462966918945, "sampling/sampling_logp_difference/mean": 0.041879359632730484, "step": 1290 }, { "clip_ratio/high_max": 0.024315623799338937, "clip_ratio/high_mean": 0.024315623799338937, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/region_mean": 0.030565623892471194, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 102.5, "completions/mean_terminated_length": 102.5, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.21113021485507488, "epoch": 0.051853636984375624, "frac_reward_zero_std": 0.0, "grad_norm": 3.836810350418091, "learning_rate": 6.090909090909092e-06, "loss": -0.0041, "num_tokens": 2918069.0, "reward": 0.9708477258682251, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957513809204102, "reward_meter_std": 0.0030371833126991987, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07034352421760559, "reward_total_composite_mean": 0.9708477258682251, "reward_total_composite_std": 0.07034352421760559, "reward_total_mean": 0.9708477258682251, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957513809204102, "rewards/meter/std": 0.0030371833126991987, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9708477258682251, "rewards/total_composite/std": 0.07034352421760559, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0092651844024658, "sampling/importance_sampling_ratio/min": 0.4889427423477173, "sampling/sampling_logp_difference/max": 0.7555546760559082, "sampling/sampling_logp_difference/mean": 0.03093818947672844, "step": 1291 }, { "clip_ratio/high_max": 0.03230128774885088, "clip_ratio/high_mean": 0.03230128774885088, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/region_mean": 0.03905804466921836, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.4877397455275059, "epoch": 0.05189380246616058, "frac_reward_zero_std": 0.0, "grad_norm": 4.021605968475342, "learning_rate": 6.087878787878788e-06, "loss": 0.0073, "num_tokens": 2919873.0, "reward": 0.9728573560714722, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9728573560714722, "reward_meter_std": 0.0436580516397953, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0436580590903759, "reward_total_composite_mean": 0.9728573560714722, "reward_total_composite_std": 0.0436580516397953, "reward_total_mean": 0.9728573560714722, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9728573560714722, "rewards/meter/std": 0.0436580516397953, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9728573560714722, "rewards/total_composite/std": 0.0436580516397953, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0169715881347656, "sampling/importance_sampling_ratio/min": 0.23279628157615662, "sampling/sampling_logp_difference/max": 1.4575915336608887, "sampling/sampling_logp_difference/mean": 0.06107901781797409, "step": 1292 }, { "clip_ratio/high_max": 0.027407341869547963, "clip_ratio/high_mean": 0.027407341869547963, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.027407341869547963, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 37.125, "completions/mean_terminated_length": 37.125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.13870495092123747, "epoch": 0.05193396794794554, "frac_reward_zero_std": 0.0, "grad_norm": 6.868655681610107, "learning_rate": 6.0848484848484855e-06, "loss": 0.0401, "num_tokens": 2921458.0, "reward": 0.9967739582061768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9967739582061768, "reward_meter_std": 0.002069387584924698, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00206937943585217, "reward_total_composite_mean": 0.9967739582061768, "reward_total_composite_std": 0.002069387584924698, "reward_total_mean": 0.9967739582061768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9967739582061768, "rewards/meter/std": 0.002069387584924698, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967739582061768, "rewards/total_composite/std": 0.002069387584924698, "sampling/importance_sampling_ratio/max": 1.8224308490753174, "sampling/importance_sampling_ratio/mean": 0.9899824857711792, "sampling/importance_sampling_ratio/min": 0.11321556568145752, "sampling/sampling_logp_difference/max": 2.1784615516662598, "sampling/sampling_logp_difference/mean": 0.04344024881720543, "step": 1293 }, { "clip_ratio/high_max": 0.031401457847096026, "clip_ratio/high_mean": 0.031401457847096026, "clip_ratio/low_mean": 0.005599473137408495, "clip_ratio/low_min": 0.005599473137408495, "clip_ratio/region_mean": 0.03700093098450452, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 67.5, "completions/mean_terminated_length": 67.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.30966538563370705, "epoch": 0.05197413342973049, "frac_reward_zero_std": 0.0, "grad_norm": 4.397799491882324, "learning_rate": 6.081818181818182e-06, "loss": 0.0191, "num_tokens": 2923230.0, "reward": 0.938705325126648, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.938705325126648, "reward_meter_std": 0.06307699531316757, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06307698041200638, "reward_total_composite_mean": 0.938705325126648, "reward_total_composite_std": 0.06307699531316757, "reward_total_mean": 0.938705325126648, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.938705325126648, "rewards/meter/std": 0.06307699531316757, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.938705325126648, "rewards/total_composite/std": 0.06307699531316757, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0048556327819824, "sampling/importance_sampling_ratio/min": 0.11140039563179016, "sampling/sampling_logp_difference/max": 2.194624423980713, "sampling/sampling_logp_difference/mean": 0.05176489055156708, "step": 1294 }, { "clip_ratio/high_max": 0.032499466673471034, "clip_ratio/high_mean": 0.032499466673471034, "clip_ratio/low_mean": 0.011194029822945595, "clip_ratio/low_min": 0.011194029822945595, "clip_ratio/region_mean": 0.04369349649641663, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.3127963822335005, "epoch": 0.05201429891151545, "frac_reward_zero_std": 0.0, "grad_norm": 7.243402481079102, "learning_rate": 6.07878787878788e-06, "loss": 0.0019, "num_tokens": 2924989.0, "reward": 0.9284621477127075, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9284621477127075, "reward_meter_std": 0.13208235800266266, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.13208235800266266, "reward_total_composite_mean": 0.9284621477127075, "reward_total_composite_std": 0.13208235800266266, "reward_total_mean": 0.9284621477127075, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9284621477127075, "rewards/meter/std": 0.13208235800266266, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9284621477127075, "rewards/total_composite/std": 0.13208235800266266, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0104353427886963, "sampling/importance_sampling_ratio/min": 0.33756616711616516, "sampling/sampling_logp_difference/max": 1.085993766784668, "sampling/sampling_logp_difference/mean": 0.05147900432348251, "step": 1295 }, { "clip_ratio/high_max": 0.04434093181043863, "clip_ratio/high_mean": 0.04434093181043863, "clip_ratio/low_mean": 0.01175213698297739, "clip_ratio/low_min": 0.01175213698297739, "clip_ratio/region_mean": 0.05609306879341602, "completions/clipped_ratio": 0.0, "completions/max_length": 115.0, "completions/max_terminated_length": 115.0, "completions/mean_length": 108.25, "completions/mean_terminated_length": 108.25, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.6056771166622639, "epoch": 0.0520544643933004, "frac_reward_zero_std": 0.0, "grad_norm": 5.327840328216553, "learning_rate": 6.0757575757575755e-06, "loss": 0.0071, "num_tokens": 2927167.0, "reward": 0.8149079084396362, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8149079084396362, "reward_meter_std": 0.24847427010536194, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.24847425520420074, "reward_total_composite_mean": 0.8149079084396362, "reward_total_composite_std": 0.24847427010536194, "reward_total_mean": 0.8149079084396362, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8149079084396362, "rewards/meter/std": 0.24847427010536194, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8149079084396362, "rewards/total_composite/std": 0.24847427010536194, "sampling/importance_sampling_ratio/max": 1.9841198921203613, "sampling/importance_sampling_ratio/mean": 1.0199620723724365, "sampling/importance_sampling_ratio/min": 0.20831067860126495, "sampling/sampling_logp_difference/max": 1.5687246322631836, "sampling/sampling_logp_difference/mean": 0.06685814261436462, "step": 1296 }, { "clip_ratio/high_max": 0.02069655992090702, "clip_ratio/high_mean": 0.02069655992090702, "clip_ratio/low_mean": 0.015135752153582871, "clip_ratio/low_min": 0.015135752153582871, "clip_ratio/region_mean": 0.03583231207448989, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 73.125, "completions/mean_terminated_length": 73.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.3857220932841301, "epoch": 0.052094629875085355, "frac_reward_zero_std": 0.0, "grad_norm": 4.677238941192627, "learning_rate": 6.072727272727274e-06, "loss": 0.019, "num_tokens": 2929000.0, "reward": 0.9951352477073669, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9951352477073669, "reward_meter_std": 0.0018285271944478154, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0018285303376615047, "reward_total_composite_mean": 0.9951352477073669, "reward_total_composite_std": 0.0018285271944478154, "reward_total_mean": 0.9951352477073669, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9951352477073669, "rewards/meter/std": 0.0018285271944478154, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951352477073669, "rewards/total_composite/std": 0.0018285271944478154, "sampling/importance_sampling_ratio/max": 1.4575989246368408, "sampling/importance_sampling_ratio/mean": 1.010929822921753, "sampling/importance_sampling_ratio/min": 0.1286551058292389, "sampling/sampling_logp_difference/max": 2.0506200790405273, "sampling/sampling_logp_difference/mean": 0.0533045269548893, "step": 1297 }, { "clip_ratio/high_max": 0.02205882384441793, "clip_ratio/high_mean": 0.02205882384441793, "clip_ratio/low_mean": 0.014928699005395174, "clip_ratio/low_min": 0.014928699005395174, "clip_ratio/region_mean": 0.036987522849813104, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.0, "completions/mean_terminated_length": 34.0, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.4743592441082001, "epoch": 0.05213479535687031, "frac_reward_zero_std": 0.0, "grad_norm": 7.133460521697998, "learning_rate": 6.06969696969697e-06, "loss": 0.0077, "num_tokens": 2930480.0, "reward": 0.9618321657180786, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9618321657180786, "reward_meter_std": 0.022519638761878014, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.022519640624523163, "reward_total_composite_mean": 0.9618321657180786, "reward_total_composite_std": 0.022519638761878014, "reward_total_mean": 0.9618321657180786, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9618321657180786, "rewards/meter/std": 0.022519638761878014, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9618321657180786, "rewards/total_composite/std": 0.022519638761878014, "sampling/importance_sampling_ratio/max": 1.5166521072387695, "sampling/importance_sampling_ratio/mean": 1.023314118385315, "sampling/importance_sampling_ratio/min": 0.4840027689933777, "sampling/sampling_logp_difference/max": 0.7256646156311035, "sampling/sampling_logp_difference/mean": 0.04984576627612114, "step": 1298 }, { "clip_ratio/high_max": 0.02266476070508361, "clip_ratio/high_mean": 0.02266476070508361, "clip_ratio/low_mean": 0.04831210756674409, "clip_ratio/low_min": 0.04831210756674409, "clip_ratio/region_mean": 0.0709768682718277, "completions/clipped_ratio": 0.0, "completions/max_length": 97.0, "completions/max_terminated_length": 97.0, "completions/mean_length": 91.0, "completions/mean_terminated_length": 91.0, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.7791374623775482, "epoch": 0.05217496083865526, "frac_reward_zero_std": 0.0, "grad_norm": 6.881636142730713, "learning_rate": 6.066666666666667e-06, "loss": -0.0198, "num_tokens": 2932528.0, "reward": 0.36679619550704956, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.37875261902809143, "reward_meter_std": 0.31527000665664673, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.3127610683441162, "reward_total_composite_mean": 0.36679619550704956, "reward_total_composite_std": 0.3127610683441162, "reward_total_mean": 0.36679619550704956, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.37875261902809143, "rewards/meter/std": 0.31527000665664673, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.36679619550704956, "rewards/total_composite/std": 0.3127610683441162, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0225510597229004, "sampling/importance_sampling_ratio/min": 0.2555953860282898, "sampling/sampling_logp_difference/max": 1.3641595840454102, "sampling/sampling_logp_difference/mean": 0.0896112248301506, "step": 1299 }, { "clip_ratio/high_max": 0.04649715404957533, "clip_ratio/high_mean": 0.04649715404957533, "clip_ratio/low_mean": 0.016562293516471982, "clip_ratio/low_min": 0.016562293516471982, "clip_ratio/region_mean": 0.06305944756604731, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 38.0, "completions/mean_terminated_length": 38.0, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.4784490168094635, "epoch": 0.05221512632044022, "frac_reward_zero_std": 0.0, "grad_norm": 10.192344665527344, "learning_rate": 6.063636363636364e-06, "loss": 0.0288, "num_tokens": 2933984.0, "reward": 0.6407955884933472, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6407955884933472, "reward_meter_std": 0.46960803866386414, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.46960803866386414, "reward_total_composite_mean": 0.6407955884933472, "reward_total_composite_std": 0.46960803866386414, "reward_total_mean": 0.6407955884933472, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6407955884933472, "rewards/meter/std": 0.46960803866386414, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6407955884933472, "rewards/total_composite/std": 0.46960803866386414, "sampling/importance_sampling_ratio/max": 1.5791682004928589, "sampling/importance_sampling_ratio/mean": 1.007232427597046, "sampling/importance_sampling_ratio/min": 0.21284130215644836, "sampling/sampling_logp_difference/max": 1.5472084283828735, "sampling/sampling_logp_difference/mean": 0.07035049796104431, "step": 1300 }, { "epoch": 0.05221512632044022, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 285.9230769230769, "eval_completions/max_terminated_length": 285.9230769230769, "eval_completions/mean_length": 170.44230769230768, "eval_completions/mean_terminated_length": 170.44230769230768, "eval_completions/min_length": 62.07692307692308, "eval_completions/min_terminated_length": 62.07692307692308, "eval_entropy": 0.24410619873266953, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 2933984.0, "eval_reward": 0.36767612856168014, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.8285529888593234, "eval_reward_count_adherence_std": 0.17163905444053504, "eval_reward_meter_mean": 0.5045671004515427, "eval_reward_meter_std": 0.4269758417056157, "eval_reward_repeat_penalty_mean": 0.8525567375696622, "eval_reward_repeat_penalty_std": 0.13138744378319153, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.36767612856168014, "eval_reward_total_composite_std": 0.36054489933527434, "eval_reward_total_mean": 0.36767612856168014, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.8285529888593234, "eval_rewards/count_adherence/std": 0.17163905444053504, "eval_rewards/meter/mean": 0.5045671004515427, "eval_rewards/meter/std": 0.4269758417056157, "eval_rewards/repeat_penalty/mean": 0.8525567375696622, "eval_rewards/repeat_penalty/std": 0.13138744378319153, "eval_rewards/total_composite/mean": 0.36767612856168014, "eval_rewards/total_composite/std": 0.36054489933527434, "eval_runtime": 55.9753, "eval_samples_per_second": 1.858, "eval_sampling/importance_sampling_ratio/max": 1.4160877099403968, "eval_sampling/importance_sampling_ratio/mean": 1.0066082569269033, "eval_sampling/importance_sampling_ratio/min": 0.35662598334825957, "eval_sampling/sampling_logp_difference/max": 1.0623591863192046, "eval_sampling/sampling_logp_difference/mean": 0.02532697569292325, "eval_steps_per_second": 0.232, "step": 1300 }, { "clip_ratio/high_max": 0.03823378193192184, "clip_ratio/high_mean": 0.03823378193192184, "clip_ratio/low_mean": 0.010581943904981017, "clip_ratio/low_min": 0.010581943904981017, "clip_ratio/region_mean": 0.04881572583690286, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.875, "completions/mean_terminated_length": 71.875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.47279578633606434, "epoch": 0.05225529180222517, "frac_reward_zero_std": 0.0, "grad_norm": 5.742159843444824, "learning_rate": 6.060606060606061e-06, "loss": 0.0021, "num_tokens": 2935839.0, "reward": 0.5558804869651794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5558804869651794, "reward_meter_std": 0.43805626034736633, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.43805623054504395, "reward_total_composite_mean": 0.5558804869651794, "reward_total_composite_std": 0.43805626034736633, "reward_total_mean": 0.5558804869651794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5558804869651794, "rewards/meter/std": 0.43805626034736633, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5558804869651794, "rewards/total_composite/std": 0.43805626034736633, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0061253309249878, "sampling/importance_sampling_ratio/min": 0.15141265094280243, "sampling/sampling_logp_difference/max": 1.8877463340759277, "sampling/sampling_logp_difference/mean": 0.061652250587940216, "step": 1301 }, { "clip_ratio/high_max": 0.005509414477273822, "clip_ratio/high_mean": 0.005509414477273822, "clip_ratio/low_mean": 0.013534625410102308, "clip_ratio/low_min": 0.013534625410102308, "clip_ratio/region_mean": 0.01904403988737613, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 322.375, "completions/mean_terminated_length": 322.375, "completions/min_length": 317.0, "completions/min_terminated_length": 317.0, "entropy": 0.1677823355421424, "epoch": 0.052295457284010124, "frac_reward_zero_std": 0.0, "grad_norm": 1.432262897491455, "learning_rate": 6.057575757575757e-06, "loss": 0.0079, "num_tokens": 2939778.0, "reward": 0.32962822914123535, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5333333611488342, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8052475452423096, "reward_meter_std": 0.0374782457947731, "reward_repeat_penalty_mean": 0.7666666507720947, "reward_repeat_penalty_std": 0.06172133609652519, "reward_std": 0.0352957658469677, "reward_total_composite_mean": 0.32962822914123535, "reward_total_composite_std": 0.0352957658469677, "reward_total_mean": 0.32962822914123535, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5333333611488342, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8052475452423096, "rewards/meter/std": 0.0374782457947731, "rewards/repeat_penalty/mean": 0.7666666507720947, "rewards/repeat_penalty/std": 0.06172133609652519, "rewards/total_composite/mean": 0.32962822914123535, "rewards/total_composite/std": 0.0352957658469677, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0048707723617554, "sampling/importance_sampling_ratio/min": 0.24125131964683533, "sampling/sampling_logp_difference/max": 1.421916127204895, "sampling/sampling_logp_difference/mean": 0.024004999548196793, "step": 1302 }, { "clip_ratio/high_max": 0.018164411187171936, "clip_ratio/high_mean": 0.018164411187171936, "clip_ratio/low_mean": 0.010266162920743227, "clip_ratio/low_min": 0.010266162920743227, "clip_ratio/region_mean": 0.028430574107915163, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 119.0, "completions/mean_terminated_length": 119.0, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.17016714438796043, "epoch": 0.05233562276579508, "frac_reward_zero_std": 0.0, "grad_norm": 3.591283082962036, "learning_rate": 6.0545454545454555e-06, "loss": 0.0168, "num_tokens": 2942114.0, "reward": 0.6865410804748535, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8989412188529968, "reward_meter_std": 0.138026162981987, "reward_repeat_penalty_mean": 0.7678571343421936, "reward_repeat_penalty_std": 0.15152288973331451, "reward_std": 0.16533119976520538, "reward_total_composite_mean": 0.6865410804748535, "reward_total_composite_std": 0.16533121466636658, "reward_total_mean": 0.6865410804748535, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8989412188529968, "rewards/meter/std": 0.138026162981987, "rewards/repeat_penalty/mean": 0.7678571343421936, "rewards/repeat_penalty/std": 0.15152288973331451, "rewards/total_composite/mean": 0.6865410804748535, "rewards/total_composite/std": 0.16533121466636658, "sampling/importance_sampling_ratio/max": 1.522925853729248, "sampling/importance_sampling_ratio/mean": 1.0041345357894897, "sampling/importance_sampling_ratio/min": 0.26649829745292664, "sampling/sampling_logp_difference/max": 1.322387456893921, "sampling/sampling_logp_difference/mean": 0.027758914977312088, "step": 1303 }, { "clip_ratio/high_max": 0.011577953351661563, "clip_ratio/high_mean": 0.011577953351661563, "clip_ratio/low_mean": 0.00956214708276093, "clip_ratio/low_min": 0.00956214708276093, "clip_ratio/region_mean": 0.021140100434422493, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 247.5, "completions/mean_terminated_length": 247.5, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.1491470541805029, "epoch": 0.05237578824758003, "frac_reward_zero_std": 0.0, "grad_norm": 3.418581962585449, "learning_rate": 6.051515151515152e-06, "loss": -0.0515, "num_tokens": 2945734.0, "reward": 0.5886770486831665, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.796875, "reward_count_adherence_std": 0.06469365209341049, "reward_meter_mean": 0.9368978142738342, "reward_meter_std": 0.09892511367797852, "reward_repeat_penalty_mean": 0.7849650382995605, "reward_repeat_penalty_std": 0.07891790568828583, "reward_std": 0.11287945508956909, "reward_total_composite_mean": 0.5886770486831665, "reward_total_composite_std": 0.11287946254014969, "reward_total_mean": 0.5886770486831665, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.796875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/meter/mean": 0.9368978142738342, "rewards/meter/std": 0.09892511367797852, "rewards/repeat_penalty/mean": 0.7849650382995605, "rewards/repeat_penalty/std": 0.07891790568828583, "rewards/total_composite/mean": 0.5886770486831665, "rewards/total_composite/std": 0.11287946254014969, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0041065216064453, "sampling/importance_sampling_ratio/min": 0.19896115362644196, "sampling/sampling_logp_difference/max": 1.6146456003189087, "sampling/sampling_logp_difference/mean": 0.024336956441402435, "step": 1304 }, { "clip_ratio/high_max": 0.02036819839850068, "clip_ratio/high_mean": 0.02036819839850068, "clip_ratio/low_mean": 0.006469378364272416, "clip_ratio/low_min": 0.006469378364272416, "clip_ratio/region_mean": 0.026837576762773097, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 132.875, "completions/mean_terminated_length": 132.875, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.15318176615983248, "epoch": 0.052415953729364986, "frac_reward_zero_std": 0.0, "grad_norm": 3.999876022338867, "learning_rate": 6.048484848484849e-06, "loss": 0.0351, "num_tokens": 2948213.0, "reward": 0.4297538995742798, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.45057880878448486, "reward_meter_std": 0.39947640895843506, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.3954527974128723, "reward_total_composite_mean": 0.4297538995742798, "reward_total_composite_std": 0.3954527974128723, "reward_total_mean": 0.4297538995742798, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.45057880878448486, "rewards/meter/std": 0.39947640895843506, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.4297538995742798, "rewards/total_composite/std": 0.3954527974128723, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0062963962554932, "sampling/importance_sampling_ratio/min": 0.2726346254348755, "sampling/sampling_logp_difference/max": 1.2996227741241455, "sampling/sampling_logp_difference/mean": 0.03056526370346546, "step": 1305 }, { "clip_ratio/high_max": 0.031160252634435892, "clip_ratio/high_mean": 0.031160252634435892, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/region_mean": 0.03428525268100202, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 38.75, "completions/mean_terminated_length": 38.75, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.4362948350608349, "epoch": 0.05245611921114994, "frac_reward_zero_std": 0.0, "grad_norm": 5.305455207824707, "learning_rate": 6.0454545454545456e-06, "loss": 0.0182, "num_tokens": 2949699.0, "reward": 0.9870239496231079, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9870239496231079, "reward_meter_std": 0.016622211784124374, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.016622209921479225, "reward_total_composite_mean": 0.9870239496231079, "reward_total_composite_std": 0.016622211784124374, "reward_total_mean": 0.9870239496231079, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9870239496231079, "rewards/meter/std": 0.016622211784124374, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9870239496231079, "rewards/total_composite/std": 0.016622211784124374, "sampling/importance_sampling_ratio/max": 1.6217622756958008, "sampling/importance_sampling_ratio/mean": 1.015398621559143, "sampling/importance_sampling_ratio/min": 0.2841854691505432, "sampling/sampling_logp_difference/max": 1.2581281661987305, "sampling/sampling_logp_difference/mean": 0.05584869906306267, "step": 1306 }, { "clip_ratio/high_max": 0.023235379019752145, "clip_ratio/high_mean": 0.023235379019752145, "clip_ratio/low_mean": 0.012753612943924963, "clip_ratio/low_min": 0.012753612943924963, "clip_ratio/region_mean": 0.03598899196367711, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 96.75, "completions/mean_terminated_length": 96.75, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.36078083515167236, "epoch": 0.052496284692934894, "frac_reward_zero_std": 0.0, "grad_norm": 4.432555198669434, "learning_rate": 6.042424242424243e-06, "loss": 0.0252, "num_tokens": 2951929.0, "reward": 0.5558610558509827, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6255627274513245, "reward_meter_std": 0.33046483993530273, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.14880475401878357, "reward_std": 0.28336387872695923, "reward_total_composite_mean": 0.5558610558509827, "reward_total_composite_std": 0.2833639085292816, "reward_total_mean": 0.5558610558509827, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6255627274513245, "rewards/meter/std": 0.33046483993530273, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.14880475401878357, "rewards/total_composite/mean": 0.5558610558509827, "rewards/total_composite/std": 0.2833639085292816, "sampling/importance_sampling_ratio/max": 1.8040260076522827, "sampling/importance_sampling_ratio/mean": 1.008318543434143, "sampling/importance_sampling_ratio/min": 0.25754326581954956, "sampling/sampling_logp_difference/max": 1.356567621231079, "sampling/sampling_logp_difference/mean": 0.048519354313611984, "step": 1307 }, { "clip_ratio/high_max": 0.021340470295399427, "clip_ratio/high_mean": 0.021340470295399427, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/region_mean": 0.025186624145135283, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 64.375, "completions/mean_terminated_length": 64.375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.11791138723492622, "epoch": 0.05253645017471985, "frac_reward_zero_std": 0.0, "grad_norm": 6.026689529418945, "learning_rate": 6.039393939393939e-06, "loss": 0.0072, "num_tokens": 2953756.0, "reward": 0.9559545516967773, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9559545516967773, "reward_meter_std": 0.06238928064703941, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.062389273196458817, "reward_total_composite_mean": 0.9559545516967773, "reward_total_composite_std": 0.06238928064703941, "reward_total_mean": 0.9559545516967773, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9559545516967773, "rewards/meter/std": 0.06238928064703941, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9559545516967773, "rewards/total_composite/std": 0.06238928064703941, "sampling/importance_sampling_ratio/max": 1.5961639881134033, "sampling/importance_sampling_ratio/mean": 1.000700831413269, "sampling/importance_sampling_ratio/min": 0.09954863786697388, "sampling/sampling_logp_difference/max": 2.3071088790893555, "sampling/sampling_logp_difference/mean": 0.031952135264873505, "step": 1308 }, { "clip_ratio/high_max": 0.030007666442543268, "clip_ratio/high_mean": 0.030007666442543268, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.03357909503392875, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.375, "completions/mean_terminated_length": 34.375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.3678742181509733, "epoch": 0.0525766156565048, "frac_reward_zero_std": 0.0, "grad_norm": 6.288991928100586, "learning_rate": 6.0363636363636365e-06, "loss": 0.0141, "num_tokens": 2955303.0, "reward": 0.8419111371040344, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8419111371040344, "reward_meter_std": 0.3378627300262451, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.33786270022392273, "reward_total_composite_mean": 0.8419111371040344, "reward_total_composite_std": 0.3378627300262451, "reward_total_mean": 0.8419111371040344, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8419111371040344, "rewards/meter/std": 0.3378627300262451, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8419111371040344, "rewards/total_composite/std": 0.3378627300262451, "sampling/importance_sampling_ratio/max": 1.8161826133728027, "sampling/importance_sampling_ratio/mean": 1.0118752717971802, "sampling/importance_sampling_ratio/min": 0.5022655725479126, "sampling/sampling_logp_difference/max": 0.6886262893676758, "sampling/sampling_logp_difference/mean": 0.03979434072971344, "step": 1309 }, { "clip_ratio/high_max": 0.0035462776431813836, "clip_ratio/high_mean": 0.0035462776431813836, "clip_ratio/low_mean": 0.010817805537953973, "clip_ratio/low_min": 0.010817805537953973, "clip_ratio/region_mean": 0.014364083181135356, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 70.125, "completions/mean_terminated_length": 70.125, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.10299927368760109, "epoch": 0.052616781138289756, "frac_reward_zero_std": 0.0, "grad_norm": 4.350625514984131, "learning_rate": 6.033333333333335e-06, "loss": -0.0014, "num_tokens": 2957280.0, "reward": 0.9980206489562988, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980206489562988, "reward_meter_std": 0.00018548252410255373, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00018549479136709124, "reward_total_composite_mean": 0.9980206489562988, "reward_total_composite_std": 0.00018548252410255373, "reward_total_mean": 0.9980206489562988, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980206489562988, "rewards/meter/std": 0.00018548252410255373, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980206489562988, "rewards/total_composite/std": 0.00018548252410255373, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0046977996826172, "sampling/importance_sampling_ratio/min": 0.3045315444469452, "sampling/sampling_logp_difference/max": 1.1889805793762207, "sampling/sampling_logp_difference/mean": 0.022730743512511253, "step": 1310 }, { "clip_ratio/high_max": 0.0035714286495931447, "clip_ratio/high_mean": 0.0035714286495931447, "clip_ratio/low_mean": 0.01160750730196014, "clip_ratio/low_min": 0.01160750730196014, "clip_ratio/region_mean": 0.015178935951553285, "completions/clipped_ratio": 0.0, "completions/max_length": 141.0, "completions/max_terminated_length": 141.0, "completions/mean_length": 140.125, "completions/mean_terminated_length": 140.125, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.11066047195345163, "epoch": 0.05265694662007471, "frac_reward_zero_std": 0.0, "grad_norm": 2.434211015701294, "learning_rate": 6.030303030303031e-06, "loss": 0.0065, "num_tokens": 2959801.0, "reward": 0.7309621572494507, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983865022659302, "reward_meter_std": 2.5610979719203897e-05, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.09155284613370895, "reward_std": 0.09140852838754654, "reward_total_composite_mean": 0.7309621572494507, "reward_total_composite_std": 0.09140855073928833, "reward_total_mean": 0.7309621572494507, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983865022659302, "rewards/meter/std": 2.5610979719203897e-05, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.7309621572494507, "rewards/total_composite/std": 0.09140855073928833, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0040990114212036, "sampling/importance_sampling_ratio/min": 0.17856921255588531, "sampling/sampling_logp_difference/max": 1.7227790355682373, "sampling/sampling_logp_difference/mean": 0.021905627101659775, "step": 1311 }, { "clip_ratio/high_max": 0.016737288795411587, "clip_ratio/high_mean": 0.016737288795411587, "clip_ratio/low_mean": 0.006250000325962901, "clip_ratio/low_min": 0.006250000325962901, "clip_ratio/region_mean": 0.022987289121374488, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 59.875, "completions/mean_terminated_length": 59.875, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.16338308528065681, "epoch": 0.052697112101859664, "frac_reward_zero_std": 0.0, "grad_norm": 5.163435459136963, "learning_rate": 6.027272727272728e-06, "loss": -0.0003, "num_tokens": 2961576.0, "reward": 0.8866318464279175, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8866318464279175, "reward_meter_std": 0.18304355442523956, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.18304355442523956, "reward_total_composite_mean": 0.8866318464279175, "reward_total_composite_std": 0.18304355442523956, "reward_total_mean": 0.8866318464279175, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8866318464279175, "rewards/meter/std": 0.18304355442523956, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8866318464279175, "rewards/total_composite/std": 0.18304355442523956, "sampling/importance_sampling_ratio/max": 1.9061437845230103, "sampling/importance_sampling_ratio/mean": 1.0048235654830933, "sampling/importance_sampling_ratio/min": 0.3728075623512268, "sampling/sampling_logp_difference/max": 0.9866929054260254, "sampling/sampling_logp_difference/mean": 0.02904520370066166, "step": 1312 }, { "clip_ratio/high_max": 0.01937754324171692, "clip_ratio/high_mean": 0.01937754324171692, "clip_ratio/low_mean": 0.010145591222681105, "clip_ratio/low_min": 0.010145591222681105, "clip_ratio/region_mean": 0.029523134464398026, "completions/clipped_ratio": 0.0, "completions/max_length": 179.0, "completions/max_terminated_length": 179.0, "completions/mean_length": 172.5, "completions/mean_terminated_length": 172.5, "completions/min_length": 168.0, "completions/min_terminated_length": 168.0, "entropy": 0.2585892491042614, "epoch": 0.05273727758364462, "frac_reward_zero_std": 0.0, "grad_norm": 3.3765552043914795, "learning_rate": 6.024242424242425e-06, "loss": 0.0116, "num_tokens": 2964564.0, "reward": 0.49107223749160767, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7292158007621765, "reward_meter_std": 0.2301097810268402, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.08266931027173996, "reward_std": 0.1431041955947876, "reward_total_composite_mean": 0.49107223749160767, "reward_total_composite_std": 0.1431042104959488, "reward_total_mean": 0.49107223749160767, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7292158007621765, "rewards/meter/std": 0.2301097810268402, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.49107223749160767, "rewards/total_composite/std": 0.1431042104959488, "sampling/importance_sampling_ratio/max": 1.8441625833511353, "sampling/importance_sampling_ratio/mean": 1.005394697189331, "sampling/importance_sampling_ratio/min": 0.20726439356803894, "sampling/sampling_logp_difference/max": 1.5737600326538086, "sampling/sampling_logp_difference/mean": 0.040278006345033646, "step": 1313 }, { "clip_ratio/high_max": 0.043520811945199966, "clip_ratio/high_mean": 0.043520811945199966, "clip_ratio/low_mean": 0.005260617821477354, "clip_ratio/low_min": 0.005260617821477354, "clip_ratio/region_mean": 0.04878142976667732, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 74.0, "completions/mean_terminated_length": 74.0, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.24265113845467567, "epoch": 0.05277744306542957, "frac_reward_zero_std": 0.0, "grad_norm": 5.761319637298584, "learning_rate": 6.021212121212122e-06, "loss": -0.0091, "num_tokens": 2966484.0, "reward": 0.7603874802589417, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7603874802589417, "reward_meter_std": 0.4266778230667114, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4266778230667114, "reward_total_composite_mean": 0.7603874802589417, "reward_total_composite_std": 0.4266778230667114, "reward_total_mean": 0.7603874802589417, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7603874802589417, "rewards/meter/std": 0.4266778230667114, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7603874802589417, "rewards/total_composite/std": 0.4266778230667114, "sampling/importance_sampling_ratio/max": 1.84821355342865, "sampling/importance_sampling_ratio/mean": 0.9964911341667175, "sampling/importance_sampling_ratio/min": 0.3072781264781952, "sampling/sampling_logp_difference/max": 1.180001974105835, "sampling/sampling_logp_difference/mean": 0.042963117361068726, "step": 1314 }, { "clip_ratio/high_max": 0.018337647430598736, "clip_ratio/high_mean": 0.018337647430598736, "clip_ratio/low_mean": 0.013646618113853037, "clip_ratio/low_min": 0.013646618113853037, "clip_ratio/region_mean": 0.03198426554445177, "completions/clipped_ratio": 0.0, "completions/max_length": 115.0, "completions/max_terminated_length": 115.0, "completions/mean_length": 109.0, "completions/mean_terminated_length": 109.0, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.3228848744183779, "epoch": 0.052817608547214526, "frac_reward_zero_std": 0.0, "grad_norm": 12.346835136413574, "learning_rate": 6.018181818181818e-06, "loss": 0.0157, "num_tokens": 2968868.0, "reward": 0.9957406520843506, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957406520843506, "reward_meter_std": 0.0016969876596704125, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0016970024444162846, "reward_total_composite_mean": 0.9957406520843506, "reward_total_composite_std": 0.0016969876596704125, "reward_total_mean": 0.9957406520843506, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957406520843506, "rewards/meter/std": 0.0016969876596704125, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957406520843506, "rewards/total_composite/std": 0.0016969876596704125, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0082180500030518, "sampling/importance_sampling_ratio/min": 0.26882100105285645, "sampling/sampling_logp_difference/max": 1.3137094974517822, "sampling/sampling_logp_difference/mean": 0.050970617681741714, "step": 1315 }, { "clip_ratio/high_max": 0.01568986615166068, "clip_ratio/high_mean": 0.01568986615166068, "clip_ratio/low_mean": 0.025765649508684874, "clip_ratio/low_min": 0.025765649508684874, "clip_ratio/region_mean": 0.041455515660345554, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 63.5, "completions/mean_terminated_length": 63.5, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.21594705432653427, "epoch": 0.05285777402899948, "frac_reward_zero_std": 0.0, "grad_norm": 5.835085868835449, "learning_rate": 6.015151515151516e-06, "loss": 0.0078, "num_tokens": 2970696.0, "reward": 0.4984230399131775, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5161525011062622, "reward_meter_std": 0.3657090961933136, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3891359865665436, "reward_total_composite_mean": 0.4984230399131775, "reward_total_composite_std": 0.3891359865665436, "reward_total_mean": 0.4984230399131775, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5161525011062622, "rewards/meter/std": 0.3657090961933136, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4984230399131775, "rewards/total_composite/std": 0.3891359865665436, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006494402885437, "sampling/importance_sampling_ratio/min": 0.12328764796257019, "sampling/sampling_logp_difference/max": 2.0932350158691406, "sampling/sampling_logp_difference/mean": 0.03714925795793533, "step": 1316 }, { "clip_ratio/high_max": 0.006602564244531095, "clip_ratio/high_mean": 0.006602564244531095, "clip_ratio/low_mean": 0.008204055950045586, "clip_ratio/low_min": 0.008204055950045586, "clip_ratio/region_mean": 0.01480662019457668, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 75.75, "completions/mean_terminated_length": 75.75, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.09577750787138939, "epoch": 0.052897939510784434, "frac_reward_zero_std": 0.0, "grad_norm": 3.077263832092285, "learning_rate": 6.012121212121213e-06, "loss": 0.0052, "num_tokens": 2972590.0, "reward": 0.26845037937164307, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2975860834121704, "reward_meter_std": 0.04687010124325752, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.040515363216400146, "reward_total_composite_mean": 0.26845037937164307, "reward_total_composite_std": 0.04051537066698074, "reward_total_mean": 0.26845037937164307, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2975860834121704, "rewards/meter/std": 0.04687010124325752, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.26845037937164307, "rewards/total_composite/std": 0.04051537066698074, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.999930739402771, "sampling/importance_sampling_ratio/min": 0.37996208667755127, "sampling/sampling_logp_difference/max": 0.9676837921142578, "sampling/sampling_logp_difference/mean": 0.02295822836458683, "step": 1317 }, { "clip_ratio/high_max": 0.012667326722294092, "clip_ratio/high_mean": 0.012667326722294092, "clip_ratio/low_mean": 0.009302935097366571, "clip_ratio/low_min": 0.009302935097366571, "clip_ratio/region_mean": 0.021970261819660664, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 51.25, "completions/mean_terminated_length": 51.25, "completions/min_length": 48.0, "completions/min_terminated_length": 48.0, "entropy": 0.268137663602829, "epoch": 0.05293810499256939, "frac_reward_zero_std": 0.0, "grad_norm": 6.030852317810059, "learning_rate": 6.00909090909091e-06, "loss": 0.0333, "num_tokens": 2974432.0, "reward": 0.813197135925293, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.813197135925293, "reward_meter_std": 0.24947111308574677, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.24947108328342438, "reward_total_composite_mean": 0.813197135925293, "reward_total_composite_std": 0.24947111308574677, "reward_total_mean": 0.813197135925293, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.813197135925293, "rewards/meter/std": 0.24947111308574677, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.813197135925293, "rewards/total_composite/std": 0.24947111308574677, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0154815912246704, "sampling/importance_sampling_ratio/min": 0.26231852173805237, "sampling/sampling_logp_difference/max": 1.33819580078125, "sampling/sampling_logp_difference/mean": 0.047421231865882874, "step": 1318 }, { "clip_ratio/high_max": 0.012993455864489079, "clip_ratio/high_mean": 0.012993455864489079, "clip_ratio/low_mean": 0.006019041407853365, "clip_ratio/low_min": 0.006019041407853365, "clip_ratio/region_mean": 0.019012497272342443, "completions/clipped_ratio": 0.0, "completions/max_length": 267.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 256.125, "completions/mean_terminated_length": 256.125, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.14165427815169096, "epoch": 0.05297827047435434, "frac_reward_zero_std": 0.0, "grad_norm": 2.1162424087524414, "learning_rate": 6.0060606060606065e-06, "loss": 0.0064, "num_tokens": 2978169.0, "reward": 0.586740255355835, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.859375, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9216792583465576, "reward_meter_std": 0.17728787660598755, "reward_repeat_penalty_mean": 0.7475961446762085, "reward_repeat_penalty_std": 0.0792488381266594, "reward_std": 0.11694768071174622, "reward_total_composite_mean": 0.586740255355835, "reward_total_composite_std": 0.11694769561290741, "reward_total_mean": 0.586740255355835, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.859375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9216792583465576, "rewards/meter/std": 0.17728787660598755, "rewards/repeat_penalty/mean": 0.7475961446762085, "rewards/repeat_penalty/std": 0.0792488381266594, "rewards/total_composite/mean": 0.586740255355835, "rewards/total_composite/std": 0.11694769561290741, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0010830163955688, "sampling/importance_sampling_ratio/min": 0.12540560960769653, "sampling/sampling_logp_difference/max": 2.076201915740967, "sampling/sampling_logp_difference/mean": 0.026936307549476624, "step": 1319 }, { "clip_ratio/high_max": 0.03416440007276833, "clip_ratio/high_mean": 0.03416440007276833, "clip_ratio/low_mean": 0.007752403849735856, "clip_ratio/low_min": 0.007752403849735856, "clip_ratio/region_mean": 0.04191680392250419, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 63.375, "completions/mean_terminated_length": 63.375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.4745039828121662, "epoch": 0.053018435956139295, "frac_reward_zero_std": 0.0, "grad_norm": 6.174464225769043, "learning_rate": 6.003030303030304e-06, "loss": 0.0292, "num_tokens": 2980004.0, "reward": 0.6976158022880554, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6976158022880554, "reward_meter_std": 0.3385031819343567, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3385031521320343, "reward_total_composite_mean": 0.6976158022880554, "reward_total_composite_std": 0.3385031819343567, "reward_total_mean": 0.6976158022880554, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6976158022880554, "rewards/meter/std": 0.3385031819343567, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6976158022880554, "rewards/total_composite/std": 0.3385031819343567, "sampling/importance_sampling_ratio/max": 1.8571363687515259, "sampling/importance_sampling_ratio/mean": 1.0181591510772705, "sampling/importance_sampling_ratio/min": 0.22829218208789825, "sampling/sampling_logp_difference/max": 1.4771289825439453, "sampling/sampling_logp_difference/mean": 0.06314612179994583, "step": 1320 }, { "clip_ratio/high_max": 0.026141698006540537, "clip_ratio/high_mean": 0.026141698006540537, "clip_ratio/low_mean": 0.008787594037130475, "clip_ratio/low_min": 0.008787594037130475, "clip_ratio/region_mean": 0.03492929204367101, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 72.875, "completions/mean_terminated_length": 72.875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.1369091346859932, "epoch": 0.05305860143792425, "frac_reward_zero_std": 0.0, "grad_norm": 4.180057525634766, "learning_rate": 6e-06, "loss": 0.0145, "num_tokens": 2981843.0, "reward": 0.8855767846107483, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8855767846107483, "reward_meter_std": 0.09589327871799469, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09589327871799469, "reward_total_composite_mean": 0.8855767846107483, "reward_total_composite_std": 0.09589327871799469, "reward_total_mean": 0.8855767846107483, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8855767846107483, "rewards/meter/std": 0.09589327871799469, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8855767846107483, "rewards/total_composite/std": 0.09589327871799469, "sampling/importance_sampling_ratio/max": 1.6174578666687012, "sampling/importance_sampling_ratio/mean": 0.9989095330238342, "sampling/importance_sampling_ratio/min": 0.13731785118579865, "sampling/sampling_logp_difference/max": 1.985456943511963, "sampling/sampling_logp_difference/mean": 0.03412603959441185, "step": 1321 }, { "clip_ratio/high_max": 0.009073751280084252, "clip_ratio/high_mean": 0.009073751280084252, "clip_ratio/low_mean": 0.01510935788974166, "clip_ratio/low_min": 0.01510935788974166, "clip_ratio/region_mean": 0.02418310916982591, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 41.375, "completions/mean_terminated_length": 41.375, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.23666546866297722, "epoch": 0.0530987669197092, "frac_reward_zero_std": 0.0, "grad_norm": 3.336163282394409, "learning_rate": 5.996969696969697e-06, "loss": -0.0035, "num_tokens": 2983446.0, "reward": 0.9977743029594421, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977743029594421, "reward_meter_std": 0.0008235168061219156, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008235144196078181, "reward_total_composite_mean": 0.9977743029594421, "reward_total_composite_std": 0.0008235168061219156, "reward_total_mean": 0.9977743029594421, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977743029594421, "rewards/meter/std": 0.0008235168061219156, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977743029594421, "rewards/total_composite/std": 0.0008235168061219156, "sampling/importance_sampling_ratio/max": 1.5313419103622437, "sampling/importance_sampling_ratio/mean": 1.0045777559280396, "sampling/importance_sampling_ratio/min": 0.2856776714324951, "sampling/sampling_logp_difference/max": 1.2528910636901855, "sampling/sampling_logp_difference/mean": 0.033282674849033356, "step": 1322 }, { "clip_ratio/high_max": 0.023659866768866777, "clip_ratio/high_mean": 0.023659866768866777, "clip_ratio/low_mean": 0.025560760172083974, "clip_ratio/low_min": 0.025560760172083974, "clip_ratio/region_mean": 0.04922062694095075, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 62.625, "completions/mean_terminated_length": 62.625, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.6359868273139, "epoch": 0.05313893240149416, "frac_reward_zero_std": 0.0, "grad_norm": 7.640705108642578, "learning_rate": 5.993939393939394e-06, "loss": 0.0011, "num_tokens": 2985219.0, "reward": 0.4877135455608368, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4877135455608368, "reward_meter_std": 0.3409081995487213, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3409081995487213, "reward_total_composite_mean": 0.4877135455608368, "reward_total_composite_std": 0.3409081995487213, "reward_total_mean": 0.4877135455608368, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4877135455608368, "rewards/meter/std": 0.3409081995487213, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4877135455608368, "rewards/total_composite/std": 0.3409081995487213, "sampling/importance_sampling_ratio/max": 1.7955073118209839, "sampling/importance_sampling_ratio/mean": 1.0144366025924683, "sampling/importance_sampling_ratio/min": 0.23357106745243073, "sampling/sampling_logp_difference/max": 1.4542689323425293, "sampling/sampling_logp_difference/mean": 0.08788689225912094, "step": 1323 }, { "clip_ratio/high_max": 0.023976000724360347, "clip_ratio/high_mean": 0.023976000724360347, "clip_ratio/low_mean": 0.006097560748457909, "clip_ratio/low_min": 0.006097560748457909, "clip_ratio/region_mean": 0.030073561472818255, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 41.25, "completions/mean_terminated_length": 41.25, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.2660053391009569, "epoch": 0.05317909788327911, "frac_reward_zero_std": 0.0, "grad_norm": 6.885532855987549, "learning_rate": 5.990909090909092e-06, "loss": 0.0016, "num_tokens": 2986901.0, "reward": 0.8787177801132202, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8787177801132202, "reward_meter_std": 0.3274337947368622, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3274337649345398, "reward_total_composite_mean": 0.8787177801132202, "reward_total_composite_std": 0.3274337947368622, "reward_total_mean": 0.8787177801132202, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8787177801132202, "rewards/meter/std": 0.3274337947368622, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8787177801132202, "rewards/total_composite/std": 0.3274337947368622, "sampling/importance_sampling_ratio/max": 1.5247844457626343, "sampling/importance_sampling_ratio/mean": 1.0028668642044067, "sampling/importance_sampling_ratio/min": 0.5185403823852539, "sampling/sampling_logp_difference/max": 0.6567374467849731, "sampling/sampling_logp_difference/mean": 0.03787970542907715, "step": 1324 }, { "clip_ratio/high_max": 0.017529349657706916, "clip_ratio/high_mean": 0.017529349657706916, "clip_ratio/low_mean": 0.014129945659078658, "clip_ratio/low_min": 0.014129945659078658, "clip_ratio/region_mean": 0.031659295316785574, "completions/clipped_ratio": 0.0, "completions/max_length": 117.0, "completions/max_terminated_length": 117.0, "completions/mean_length": 108.5, "completions/mean_terminated_length": 108.5, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.15250306809321046, "epoch": 0.053219263365064065, "frac_reward_zero_std": 0.0, "grad_norm": 3.35534405708313, "learning_rate": 5.987878787878788e-06, "loss": -0.0087, "num_tokens": 2988985.0, "reward": 0.6235833168029785, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7801897525787354, "reward_meter_std": 0.25841277837753296, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.12817399203777313, "reward_std": 0.18524421751499176, "reward_total_composite_mean": 0.6235833168029785, "reward_total_composite_std": 0.18524423241615295, "reward_total_mean": 0.6235833168029785, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7801897525787354, "rewards/meter/std": 0.25841277837753296, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.12817399203777313, "rewards/total_composite/mean": 0.6235833168029785, "rewards/total_composite/std": 0.18524423241615295, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0031850337982178, "sampling/importance_sampling_ratio/min": 0.29859381914138794, "sampling/sampling_logp_difference/max": 1.2086710929870605, "sampling/sampling_logp_difference/mean": 0.03167083114385605, "step": 1325 }, { "clip_ratio/high_max": 0.014489957015030086, "clip_ratio/high_mean": 0.014489957015030086, "clip_ratio/low_mean": 0.01110323509783484, "clip_ratio/low_min": 0.01110323509783484, "clip_ratio/region_mean": 0.025593192112864926, "completions/clipped_ratio": 0.0, "completions/max_length": 268.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 258.25, "completions/mean_terminated_length": 258.25, "completions/min_length": 245.0, "completions/min_terminated_length": 245.0, "entropy": 0.151836016215384, "epoch": 0.05325942884684902, "frac_reward_zero_std": 0.0, "grad_norm": 2.242483139038086, "learning_rate": 5.984848484848486e-06, "loss": 0.0087, "num_tokens": 2992787.0, "reward": 0.4503657817840576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5833333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9577704668045044, "reward_meter_std": 0.11232133209705353, "reward_repeat_penalty_mean": 0.807692289352417, "reward_repeat_penalty_std": 0.082234226167202, "reward_std": 0.06605512648820877, "reward_total_composite_mean": 0.4503657817840576, "reward_total_composite_std": 0.06605512648820877, "reward_total_mean": 0.4503657817840576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5833333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9577704668045044, "rewards/meter/std": 0.11232133209705353, "rewards/repeat_penalty/mean": 0.807692289352417, "rewards/repeat_penalty/std": 0.082234226167202, "rewards/total_composite/mean": 0.4503657817840576, "rewards/total_composite/std": 0.06605512648820877, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0053751468658447, "sampling/importance_sampling_ratio/min": 0.15725576877593994, "sampling/sampling_logp_difference/max": 1.849881649017334, "sampling/sampling_logp_difference/mean": 0.02666037157177925, "step": 1326 }, { "clip_ratio/high_max": 0.002672330185305327, "clip_ratio/high_mean": 0.002672330185305327, "clip_ratio/low_mean": 0.008893042104318738, "clip_ratio/low_min": 0.008893042104318738, "clip_ratio/region_mean": 0.011565372289624065, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 139.5, "completions/mean_terminated_length": 139.5, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.22298285737633705, "epoch": 0.05329959432863397, "frac_reward_zero_std": 0.0, "grad_norm": 2.6014902591705322, "learning_rate": 5.981818181818182e-06, "loss": -0.0023, "num_tokens": 2995231.0, "reward": 0.8890672922134399, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958108067512512, "reward_meter_std": 0.0007901726639829576, "reward_repeat_penalty_mean": 0.8928571343421936, "reward_repeat_penalty_std": 0.10101525485515594, "reward_std": 0.10008554905653, "reward_total_composite_mean": 0.8890672922134399, "reward_total_composite_std": 0.1000855341553688, "reward_total_mean": 0.8890672922134399, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958108067512512, "rewards/meter/std": 0.0007901726639829576, "rewards/repeat_penalty/mean": 0.8928571343421936, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.8890672922134399, "rewards/total_composite/std": 0.1000855341553688, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0085175037384033, "sampling/importance_sampling_ratio/min": 0.24528244137763977, "sampling/sampling_logp_difference/max": 1.4053449630737305, "sampling/sampling_logp_difference/mean": 0.03180097043514252, "step": 1327 }, { "clip_ratio/high_max": 0.024630822706967592, "clip_ratio/high_mean": 0.024630822706967592, "clip_ratio/low_mean": 0.01024590153247118, "clip_ratio/low_min": 0.01024590153247118, "clip_ratio/region_mean": 0.03487672423943877, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.17815150693058968, "epoch": 0.05333975981041893, "frac_reward_zero_std": 0.0, "grad_norm": 9.022930145263672, "learning_rate": 5.978787878787879e-06, "loss": 0.0114, "num_tokens": 2997079.0, "reward": 0.8425968885421753, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8425968885421753, "reward_meter_std": 0.31753626465797424, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.31753626465797424, "reward_total_composite_mean": 0.8425968885421753, "reward_total_composite_std": 0.31753626465797424, "reward_total_mean": 0.8425968885421753, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8425968885421753, "rewards/meter/std": 0.31753626465797424, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8425968885421753, "rewards/total_composite/std": 0.31753626465797424, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0045350790023804, "sampling/importance_sampling_ratio/min": 0.2584564983844757, "sampling/sampling_logp_difference/max": 1.3530278205871582, "sampling/sampling_logp_difference/mean": 0.04912829399108887, "step": 1328 }, { "clip_ratio/high_max": 0.02516206307336688, "clip_ratio/high_mean": 0.02516206307336688, "clip_ratio/low_mean": 0.006886049755848944, "clip_ratio/low_min": 0.006886049755848944, "clip_ratio/region_mean": 0.032048112829215825, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 89.75, "completions/mean_terminated_length": 89.75, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.1314184283837676, "epoch": 0.05337992529220388, "frac_reward_zero_std": 0.0, "grad_norm": 3.7103993892669678, "learning_rate": 5.975757575757576e-06, "loss": 0.0153, "num_tokens": 2999253.0, "reward": 0.5941954255104065, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7116844058036804, "reward_meter_std": 0.3912390172481537, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.12817397713661194, "reward_std": 0.35485145449638367, "reward_total_composite_mean": 0.5941954255104065, "reward_total_composite_std": 0.35485145449638367, "reward_total_mean": 0.5941954255104065, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7116844058036804, "rewards/meter/std": 0.3912390172481537, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.12817397713661194, "rewards/total_composite/mean": 0.5941954255104065, "rewards/total_composite/std": 0.35485145449638367, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0010446310043335, "sampling/importance_sampling_ratio/min": 0.4242720603942871, "sampling/sampling_logp_difference/max": 0.8573803901672363, "sampling/sampling_logp_difference/mean": 0.026401184499263763, "step": 1329 }, { "clip_ratio/high_max": 0.025107774534262717, "clip_ratio/high_mean": 0.025107774534262717, "clip_ratio/low_mean": 0.0052083334885537624, "clip_ratio/low_min": 0.0052083334885537624, "clip_ratio/region_mean": 0.03031610802281648, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 70.25, "completions/mean_terminated_length": 70.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.29124583303928375, "epoch": 0.053420090773988835, "frac_reward_zero_std": 0.0, "grad_norm": 6.578309535980225, "learning_rate": 5.972727272727274e-06, "loss": 0.0134, "num_tokens": 3001055.0, "reward": 0.8781044483184814, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8781044483184814, "reward_meter_std": 0.32534006237983704, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.32534006237983704, "reward_total_composite_mean": 0.8781044483184814, "reward_total_composite_std": 0.32534006237983704, "reward_total_mean": 0.8781044483184814, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8781044483184814, "rewards/meter/std": 0.32534006237983704, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8781044483184814, "rewards/total_composite/std": 0.32534006237983704, "sampling/importance_sampling_ratio/max": 1.5147264003753662, "sampling/importance_sampling_ratio/mean": 1.002303957939148, "sampling/importance_sampling_ratio/min": 0.21160411834716797, "sampling/sampling_logp_difference/max": 1.5530381202697754, "sampling/sampling_logp_difference/mean": 0.04965127259492874, "step": 1330 }, { "clip_ratio/high_max": 0.028128837468102574, "clip_ratio/high_mean": 0.028128837468102574, "clip_ratio/low_mean": 0.006243940908461809, "clip_ratio/low_min": 0.006243940908461809, "clip_ratio/region_mean": 0.034372778376564384, "completions/clipped_ratio": 0.0, "completions/max_length": 148.0, "completions/max_terminated_length": 148.0, "completions/mean_length": 142.0, "completions/mean_terminated_length": 142.0, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.1445506028831005, "epoch": 0.05346025625577379, "frac_reward_zero_std": 0.0, "grad_norm": 2.5885565280914307, "learning_rate": 5.96969696969697e-06, "loss": -0.0004, "num_tokens": 3003703.0, "reward": 0.5267435908317566, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8050373792648315, "reward_meter_std": 0.2827489972114563, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.08399210125207901, "reward_std": 0.19810135662555695, "reward_total_composite_mean": 0.5267435908317566, "reward_total_composite_std": 0.19810138642787933, "reward_total_mean": 0.5267435908317566, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8050373792648315, "rewards/meter/std": 0.2827489972114563, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.5267435908317566, "rewards/total_composite/std": 0.19810138642787933, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9992785453796387, "sampling/importance_sampling_ratio/min": 0.19754497706890106, "sampling/sampling_logp_difference/max": 1.6217889785766602, "sampling/sampling_logp_difference/mean": 0.032939184457063675, "step": 1331 }, { "clip_ratio/high_max": 0.01978508895263076, "clip_ratio/high_mean": 0.01978508895263076, "clip_ratio/low_mean": 0.015480266651138663, "clip_ratio/low_min": 0.015480266651138663, "clip_ratio/region_mean": 0.03526535560376942, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 134.5, "completions/mean_terminated_length": 134.5, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.2525683268904686, "epoch": 0.05350042173755874, "frac_reward_zero_std": 0.0, "grad_norm": 5.604155540466309, "learning_rate": 5.966666666666667e-06, "loss": 0.0143, "num_tokens": 3006083.0, "reward": 0.39111724495887756, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.48692476749420166, "reward_meter_std": 0.34781885147094727, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.10101525485515594, "reward_std": 0.2647561728954315, "reward_total_composite_mean": 0.39111724495887756, "reward_total_composite_std": 0.2647561728954315, "reward_total_mean": 0.39111724495887756, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.48692476749420166, "rewards/meter/std": 0.34781885147094727, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.39111724495887756, "rewards/total_composite/std": 0.2647561728954315, "sampling/importance_sampling_ratio/max": 1.928771734237671, "sampling/importance_sampling_ratio/mean": 1.007440209388733, "sampling/importance_sampling_ratio/min": 0.403021901845932, "sampling/sampling_logp_difference/max": 0.9087643623352051, "sampling/sampling_logp_difference/mean": 0.04181791469454765, "step": 1332 }, { "clip_ratio/high_max": 0.027125590480864048, "clip_ratio/high_mean": 0.027125590480864048, "clip_ratio/low_mean": 0.041454336838796735, "clip_ratio/low_min": 0.041454336838796735, "clip_ratio/region_mean": 0.06857992731966078, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 59.625, "completions/mean_terminated_length": 59.625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.5658834353089333, "epoch": 0.0535405872193437, "frac_reward_zero_std": 0.0, "grad_norm": 8.868410110473633, "learning_rate": 5.963636363636364e-06, "loss": 0.0109, "num_tokens": 3007848.0, "reward": 0.22981885075569153, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.2418718934059143, "reward_meter_std": 0.20975346863269806, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.2094106674194336, "reward_total_composite_mean": 0.22981885075569153, "reward_total_composite_std": 0.2094106674194336, "reward_total_mean": 0.22981885075569153, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.2418718934059143, "rewards/meter/std": 0.20975346863269806, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.22981885075569153, "rewards/total_composite/std": 0.2094106674194336, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0142555236816406, "sampling/importance_sampling_ratio/min": 0.3599916100502014, "sampling/sampling_logp_difference/max": 1.021674633026123, "sampling/sampling_logp_difference/mean": 0.06638707220554352, "step": 1333 }, { "clip_ratio/high_max": 0.020469399401918054, "clip_ratio/high_mean": 0.020469399401918054, "clip_ratio/low_mean": 0.01672628615051508, "clip_ratio/low_min": 0.01672628615051508, "clip_ratio/region_mean": 0.03719568555243313, "completions/clipped_ratio": 0.0, "completions/max_length": 110.0, "completions/max_terminated_length": 110.0, "completions/mean_length": 98.25, "completions/mean_terminated_length": 98.25, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.29656792618334293, "epoch": 0.05358075270112865, "frac_reward_zero_std": 0.0, "grad_norm": 4.963626384735107, "learning_rate": 5.960606060606061e-06, "loss": -0.0233, "num_tokens": 3010034.0, "reward": 0.40488189458847046, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.5051434636116028, "reward_meter_std": 0.3091416358947754, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.1414213478565216, "reward_std": 0.278901606798172, "reward_total_composite_mean": 0.40488189458847046, "reward_total_composite_std": 0.2789016366004944, "reward_total_mean": 0.40488189458847046, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.5051434636116028, "rewards/meter/std": 0.3091416358947754, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.1414213478565216, "rewards/total_composite/mean": 0.40488189458847046, "rewards/total_composite/std": 0.2789016366004944, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0055177211761475, "sampling/importance_sampling_ratio/min": 0.08610272407531738, "sampling/sampling_logp_difference/max": 2.452214241027832, "sampling/sampling_logp_difference/mean": 0.04572216421365738, "step": 1334 }, { "clip_ratio/high_max": 0.007086224795784801, "clip_ratio/high_mean": 0.007086224795784801, "clip_ratio/low_mean": 0.0008389261784031987, "clip_ratio/low_min": 0.0008389261784031987, "clip_ratio/region_mean": 0.007925150974188, "completions/clipped_ratio": 0.0, "completions/max_length": 149.0, "completions/max_terminated_length": 149.0, "completions/mean_length": 141.125, "completions/mean_terminated_length": 141.125, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.0390528803691268, "epoch": 0.053620918182913604, "frac_reward_zero_std": 0.0, "grad_norm": 4.286172866821289, "learning_rate": 5.9575757575757575e-06, "loss": 0.02, "num_tokens": 3012715.0, "reward": 0.6366074681282043, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.891250491142273, "reward_meter_std": 0.16874396800994873, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.0, "reward_std": 0.12053138017654419, "reward_total_composite_mean": 0.6366074681282043, "reward_total_composite_std": 0.12053140252828598, "reward_total_mean": 0.6366074681282043, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.891250491142273, "rewards/meter/std": 0.16874396800994873, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6366074681282043, "rewards/total_composite/std": 0.12053140252828598, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0008882284164429, "sampling/importance_sampling_ratio/min": 0.16674718260765076, "sampling/sampling_logp_difference/max": 1.791276454925537, "sampling/sampling_logp_difference/mean": 0.013281167484819889, "step": 1335 }, { "clip_ratio/high_max": 0.022171670105308294, "clip_ratio/high_mean": 0.022171670105308294, "clip_ratio/low_mean": 0.014849004452116787, "clip_ratio/low_min": 0.014849004452116787, "clip_ratio/region_mean": 0.03702067455742508, "completions/clipped_ratio": 0.0, "completions/max_length": 235.0, "completions/max_terminated_length": 235.0, "completions/mean_length": 219.25, "completions/mean_terminated_length": 219.25, "completions/min_length": 202.0, "completions/min_terminated_length": 202.0, "entropy": 0.20540771167725325, "epoch": 0.05366108366469856, "frac_reward_zero_std": 0.0, "grad_norm": 3.4981236457824707, "learning_rate": 5.954545454545455e-06, "loss": -0.0184, "num_tokens": 3016037.0, "reward": 0.7166441082954407, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.9961979389190674, "reward_meter_std": 0.0030242139473557472, "reward_repeat_penalty_mean": 0.8533653616905212, "reward_repeat_penalty_std": 0.06368338316679001, "reward_std": 0.06638932973146439, "reward_total_composite_mean": 0.7166441082954407, "reward_total_composite_std": 0.06638932228088379, "reward_total_mean": 0.7166441082954407, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.9961979389190674, "rewards/meter/std": 0.0030242139473557472, "rewards/repeat_penalty/mean": 0.8533653616905212, "rewards/repeat_penalty/std": 0.06368338316679001, "rewards/total_composite/mean": 0.7166441082954407, "rewards/total_composite/std": 0.06638932228088379, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017598867416382, "sampling/importance_sampling_ratio/min": 0.2674316465854645, "sampling/sampling_logp_difference/max": 1.3188912868499756, "sampling/sampling_logp_difference/mean": 0.03909594938158989, "step": 1336 }, { "clip_ratio/high_max": 0.007067535363603383, "clip_ratio/high_mean": 0.007067535363603383, "clip_ratio/low_mean": 0.004836487962165847, "clip_ratio/low_min": 0.004836487962165847, "clip_ratio/region_mean": 0.01190402332576923, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 292.375, "completions/mean_terminated_length": 292.375, "completions/min_length": 277.0, "completions/min_terminated_length": 277.0, "entropy": 0.10823496524244547, "epoch": 0.05370124914648351, "frac_reward_zero_std": 0.0, "grad_norm": 1.7854704856872559, "learning_rate": 5.951515151515151e-06, "loss": -0.0154, "num_tokens": 3020248.0, "reward": 0.3849773406982422, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5500000715255737, "reward_count_adherence_std": 0.030860668048262596, "reward_meter_mean": 0.9976522922515869, "reward_meter_std": 0.0010374977719038725, "reward_repeat_penalty_mean": 0.7024509906768799, "reward_repeat_penalty_std": 0.09470956027507782, "reward_std": 0.05099942535161972, "reward_total_composite_mean": 0.3849773406982422, "reward_total_composite_std": 0.05099942535161972, "reward_total_mean": 0.3849773406982422, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5500000715255737, "rewards/count_adherence/std": 0.030860668048262596, "rewards/meter/mean": 0.9976522922515869, "rewards/meter/std": 0.0010374977719038725, "rewards/repeat_penalty/mean": 0.7024509906768799, "rewards/repeat_penalty/std": 0.09470956027507782, "rewards/total_composite/mean": 0.3849773406982422, "rewards/total_composite/std": 0.05099942535161972, "sampling/importance_sampling_ratio/max": 1.787358045578003, "sampling/importance_sampling_ratio/mean": 1.0004962682724, "sampling/importance_sampling_ratio/min": 0.16746968030929565, "sampling/sampling_logp_difference/max": 1.7869529724121094, "sampling/sampling_logp_difference/mean": 0.015756171196699142, "step": 1337 }, { "clip_ratio/high_max": 0.024901960510760546, "clip_ratio/high_mean": 0.024901960510760546, "clip_ratio/low_mean": 0.03290023095905781, "clip_ratio/low_min": 0.03290023095905781, "clip_ratio/region_mean": 0.057802191469818354, "completions/clipped_ratio": 0.0, "completions/max_length": 52.0, "completions/max_terminated_length": 52.0, "completions/mean_length": 50.0, "completions/mean_terminated_length": 50.0, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.21370963286608458, "epoch": 0.053741414628268466, "frac_reward_zero_std": 0.0, "grad_norm": 10.472454071044922, "learning_rate": 5.948484848484849e-06, "loss": 0.018, "num_tokens": 3021936.0, "reward": 0.6181282997131348, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6975492238998413, "reward_meter_std": 0.3380868434906006, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.15430334210395813, "reward_std": 0.2992437779903412, "reward_total_composite_mean": 0.6181282997131348, "reward_total_composite_std": 0.2992437779903412, "reward_total_mean": 0.6181282997131348, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6975492238998413, "rewards/meter/std": 0.3380868434906006, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.15430334210395813, "rewards/total_composite/mean": 0.6181282997131348, "rewards/total_composite/std": 0.2992437779903412, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9909482598304749, "sampling/importance_sampling_ratio/min": 0.0018369464669376612, "sampling/sampling_logp_difference/max": 6.2996506690979, "sampling/sampling_logp_difference/mean": 0.0779750868678093, "step": 1338 }, { "clip_ratio/high_max": 0.019089363981038332, "clip_ratio/high_mean": 0.019089363981038332, "clip_ratio/low_mean": 0.01771387830376625, "clip_ratio/low_min": 0.01771387830376625, "clip_ratio/region_mean": 0.03680324228480458, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 115.75, "completions/mean_terminated_length": 115.75, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.13537064753472805, "epoch": 0.05378158011005342, "frac_reward_zero_std": 0.0, "grad_norm": 2.900134325027466, "learning_rate": 5.9454545454545465e-06, "loss": 0.0, "num_tokens": 3024310.0, "reward": 0.7711630463600159, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.981162428855896, "reward_meter_std": 0.015516383573412895, "reward_repeat_penalty_mean": 0.7857142686843872, "reward_repeat_penalty_std": 0.1322600096464157, "reward_std": 0.1326371133327484, "reward_total_composite_mean": 0.7711630463600159, "reward_total_composite_std": 0.1326371133327484, "reward_total_mean": 0.7711630463600159, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.981162428855896, "rewards/meter/std": 0.015516383573412895, "rewards/repeat_penalty/mean": 0.7857142686843872, "rewards/repeat_penalty/std": 0.1322600096464157, "rewards/total_composite/mean": 0.7711630463600159, "rewards/total_composite/std": 0.1326371133327484, "sampling/importance_sampling_ratio/max": 1.8718475103378296, "sampling/importance_sampling_ratio/mean": 0.9987893104553223, "sampling/importance_sampling_ratio/min": 0.15638993680477142, "sampling/sampling_logp_difference/max": 1.8554028272628784, "sampling/sampling_logp_difference/mean": 0.030475087463855743, "step": 1339 }, { "clip_ratio/high_max": 0.020700859487988055, "clip_ratio/high_mean": 0.020700859487988055, "clip_ratio/low_mean": 0.01111179729923606, "clip_ratio/low_min": 0.01111179729923606, "clip_ratio/region_mean": 0.031812656787224114, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 90.375, "completions/mean_terminated_length": 90.375, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.16755217872560024, "epoch": 0.053821745591838374, "frac_reward_zero_std": 0.0, "grad_norm": 2.7907607555389404, "learning_rate": 5.942424242424243e-06, "loss": 0.0012, "num_tokens": 3026369.0, "reward": 0.7857738733291626, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9822173118591309, "reward_meter_std": 0.01175969373434782, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009407754056155682, "reward_total_composite_mean": 0.7857738733291626, "reward_total_composite_std": 0.009407748468220234, "reward_total_mean": 0.7857738733291626, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9822173118591309, "rewards/meter/std": 0.01175969373434782, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7857738733291626, "rewards/total_composite/std": 0.009407748468220234, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0027801990509033, "sampling/importance_sampling_ratio/min": 0.27531561255455017, "sampling/sampling_logp_difference/max": 1.289837121963501, "sampling/sampling_logp_difference/mean": 0.031274572014808655, "step": 1340 }, { "clip_ratio/high_max": 0.01639674766920507, "clip_ratio/high_mean": 0.01639674766920507, "clip_ratio/low_mean": 0.010416667209938169, "clip_ratio/low_min": 0.010416667209938169, "clip_ratio/region_mean": 0.026813414879143238, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 60.625, "completions/mean_terminated_length": 60.625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.17691593896597624, "epoch": 0.05386191107362333, "frac_reward_zero_std": 0.0, "grad_norm": 7.174100399017334, "learning_rate": 5.93939393939394e-06, "loss": 0.0155, "num_tokens": 3028190.0, "reward": 0.985583484172821, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.985583484172821, "reward_meter_std": 0.008517028763890266, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008517015725374222, "reward_total_composite_mean": 0.985583484172821, "reward_total_composite_std": 0.008517028763890266, "reward_total_mean": 0.985583484172821, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.985583484172821, "rewards/meter/std": 0.008517028763890266, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.985583484172821, "rewards/total_composite/std": 0.008517028763890266, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005619764328003, "sampling/importance_sampling_ratio/min": 0.28036075830459595, "sampling/sampling_logp_difference/max": 1.27167809009552, "sampling/sampling_logp_difference/mean": 0.030380286276340485, "step": 1341 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.003597308532334864, "clip_ratio/low_min": 0.003597308532334864, "clip_ratio/region_mean": 0.005462980130687356, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.03910455713048577, "epoch": 0.05390207655540828, "frac_reward_zero_std": 0.0, "grad_norm": 4.300682544708252, "learning_rate": 5.936363636363637e-06, "loss": 0.0171, "num_tokens": 3030079.0, "reward": 0.963999330997467, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.963999330997467, "reward_meter_std": 0.005599203985184431, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00559920072555542, "reward_total_composite_mean": 0.963999330997467, "reward_total_composite_std": 0.005599203985184431, "reward_total_mean": 0.963999330997467, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.963999330997467, "rewards/meter/std": 0.005599203985184431, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.963999330997467, "rewards/total_composite/std": 0.005599203985184431, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0003573894500732, "sampling/importance_sampling_ratio/min": 0.4147241413593292, "sampling/sampling_logp_difference/max": 0.8801417350769043, "sampling/sampling_logp_difference/mean": 0.011351129971444607, "step": 1342 }, { "clip_ratio/high_max": 0.013258611783385277, "clip_ratio/high_mean": 0.013258611783385277, "clip_ratio/low_mean": 0.01829817472025752, "clip_ratio/low_min": 0.01829817472025752, "clip_ratio/region_mean": 0.0315567865036428, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 66.25, "completions/mean_terminated_length": 66.25, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.24532775953412056, "epoch": 0.053942242037193236, "frac_reward_zero_std": 0.0, "grad_norm": 3.7703700065612793, "learning_rate": 5.933333333333335e-06, "loss": 0.029, "num_tokens": 3031857.0, "reward": 0.4224083423614502, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4224083423614502, "reward_meter_std": 0.23921163380146027, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.23921160399913788, "reward_total_composite_mean": 0.4224083423614502, "reward_total_composite_std": 0.23921163380146027, "reward_total_mean": 0.4224083423614502, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4224083423614502, "rewards/meter/std": 0.23921163380146027, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4224083423614502, "rewards/total_composite/std": 0.23921163380146027, "sampling/importance_sampling_ratio/max": 1.9895342588424683, "sampling/importance_sampling_ratio/mean": 0.9989436268806458, "sampling/importance_sampling_ratio/min": 0.0285919439047575, "sampling/sampling_logp_difference/max": 3.5546302795410156, "sampling/sampling_logp_difference/mean": 0.059910353273153305, "step": 1343 }, { "clip_ratio/high_max": 0.033956656232476234, "clip_ratio/high_mean": 0.033956656232476234, "clip_ratio/low_mean": 0.01140083302743733, "clip_ratio/low_min": 0.01140083302743733, "clip_ratio/region_mean": 0.045357489259913564, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 140.125, "completions/mean_terminated_length": 140.125, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "entropy": 0.352537851780653, "epoch": 0.05398240751897819, "frac_reward_zero_std": 0.0, "grad_norm": 6.5239176750183105, "learning_rate": 5.93030303030303e-06, "loss": 0.0095, "num_tokens": 3034322.0, "reward": 0.84794020652771, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.900975227355957, "reward_meter_std": 0.25265270471572876, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.24164395034313202, "reward_total_composite_mean": 0.84794020652771, "reward_total_composite_std": 0.24164395034313202, "reward_total_mean": 0.84794020652771, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.900975227355957, "rewards/meter/std": 0.25265270471572876, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.84794020652771, "rewards/total_composite/std": 0.24164395034313202, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0002692937850952, "sampling/importance_sampling_ratio/min": 0.17647425830364227, "sampling/sampling_logp_difference/max": 1.7345802783966064, "sampling/sampling_logp_difference/mean": 0.054831597954034805, "step": 1344 }, { "clip_ratio/high_max": 0.01858155010268092, "clip_ratio/high_mean": 0.01858155010268092, "clip_ratio/low_mean": 0.0190677959471941, "clip_ratio/low_min": 0.0190677959471941, "clip_ratio/region_mean": 0.03764934604987502, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 60.0, "completions/mean_terminated_length": 60.0, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.20175675675272942, "epoch": 0.054022573000763144, "frac_reward_zero_std": 0.0, "grad_norm": 3.739021062850952, "learning_rate": 5.927272727272728e-06, "loss": -0.0074, "num_tokens": 3036050.0, "reward": 0.9513925313949585, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9513925313949585, "reward_meter_std": 0.06354161351919174, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06354160606861115, "reward_total_composite_mean": 0.9513925313949585, "reward_total_composite_std": 0.06354161351919174, "reward_total_mean": 0.9513925313949585, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9513925313949585, "rewards/meter/std": 0.06354161351919174, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9513925313949585, "rewards/total_composite/std": 0.06354161351919174, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9930896162986755, "sampling/importance_sampling_ratio/min": 0.25428733229637146, "sampling/sampling_logp_difference/max": 1.3692903518676758, "sampling/sampling_logp_difference/mean": 0.040932025760412216, "step": 1345 }, { "clip_ratio/high_max": 0.011403630953282118, "clip_ratio/high_mean": 0.011403630953282118, "clip_ratio/low_mean": 0.0009328357991762459, "clip_ratio/low_min": 0.0009328357991762459, "clip_ratio/region_mean": 0.012336466752458364, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 132.125, "completions/mean_terminated_length": 132.125, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.04485098086297512, "epoch": 0.0540627384825481, "frac_reward_zero_std": 0.0, "grad_norm": 3.906574249267578, "learning_rate": 5.924242424242425e-06, "loss": 0.0129, "num_tokens": 3038243.0, "reward": 0.6613233089447021, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.902446985244751, "reward_meter_std": 0.14302004873752594, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.11659382283687592, "reward_total_composite_mean": 0.6613233089447021, "reward_total_composite_std": 0.11659383773803711, "reward_total_mean": 0.6613233089447021, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.902446985244751, "rewards/meter/std": 0.14302004873752594, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.6613233089447021, "rewards/total_composite/std": 0.11659383773803711, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0032755136489868, "sampling/importance_sampling_ratio/min": 0.3424758315086365, "sampling/sampling_logp_difference/max": 1.1266558170318604, "sampling/sampling_logp_difference/mean": 0.013300405815243721, "step": 1346 }, { "clip_ratio/high_max": 0.024769189301878214, "clip_ratio/high_mean": 0.024769189301878214, "clip_ratio/low_mean": 0.010106078116223216, "clip_ratio/low_min": 0.010106078116223216, "clip_ratio/region_mean": 0.03487526741810143, "completions/clipped_ratio": 0.0, "completions/max_length": 112.0, "completions/max_terminated_length": 112.0, "completions/mean_length": 111.0, "completions/mean_terminated_length": 111.0, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.27726688235998154, "epoch": 0.05410290396433305, "frac_reward_zero_std": 0.0, "grad_norm": 3.224806070327759, "learning_rate": 5.921212121212122e-06, "loss": 0.0011, "num_tokens": 3040443.0, "reward": 0.9197216033935547, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9691954851150513, "reward_meter_std": 0.036640021950006485, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.08622132986783981, "reward_total_composite_mean": 0.9197216033935547, "reward_total_composite_std": 0.08622133731842041, "reward_total_mean": 0.9197216033935547, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9691954851150513, "rewards/meter/std": 0.036640021950006485, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9197216033935547, "rewards/total_composite/std": 0.08622133731842041, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.011066198348999, "sampling/importance_sampling_ratio/min": 0.2556380033493042, "sampling/sampling_logp_difference/max": 1.3639929294586182, "sampling/sampling_logp_difference/mean": 0.03579000383615494, "step": 1347 }, { "clip_ratio/high_max": 0.0280318089062348, "clip_ratio/high_mean": 0.0280318089062348, "clip_ratio/low_mean": 0.006873097387142479, "clip_ratio/low_min": 0.006873097387142479, "clip_ratio/region_mean": 0.03490490629337728, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.75, "completions/mean_terminated_length": 71.75, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.404640544205904, "epoch": 0.054143069446118006, "frac_reward_zero_std": 0.0, "grad_norm": 4.513327121734619, "learning_rate": 5.9181818181818184e-06, "loss": 0.011, "num_tokens": 3042433.0, "reward": 0.99162358045578, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99162358045578, "reward_meter_std": 0.011257669888436794, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011257672682404518, "reward_total_composite_mean": 0.99162358045578, "reward_total_composite_std": 0.011257669888436794, "reward_total_mean": 0.99162358045578, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99162358045578, "rewards/meter/std": 0.011257669888436794, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99162358045578, "rewards/total_composite/std": 0.011257669888436794, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0111957788467407, "sampling/importance_sampling_ratio/min": 0.24684081971645355, "sampling/sampling_logp_difference/max": 1.3990116119384766, "sampling/sampling_logp_difference/mean": 0.06130176782608032, "step": 1348 }, { "clip_ratio/high_max": 0.009848485118709505, "clip_ratio/high_mean": 0.009848485118709505, "clip_ratio/low_mean": 0.004213483072817326, "clip_ratio/low_min": 0.004213483072817326, "clip_ratio/region_mean": 0.01406196819152683, "completions/clipped_ratio": 0.0, "completions/max_length": 92.0, "completions/max_terminated_length": 92.0, "completions/mean_length": 89.875, "completions/mean_terminated_length": 89.875, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.10254091210663319, "epoch": 0.05418323492790296, "frac_reward_zero_std": 0.0, "grad_norm": 3.323253631591797, "learning_rate": 5.915151515151516e-06, "loss": -0.0009, "num_tokens": 3044432.0, "reward": 0.753753125667572, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9421913623809814, "reward_meter_std": 0.0664798840880394, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0531838983297348, "reward_total_composite_mean": 0.753753125667572, "reward_total_composite_std": 0.0531839095056057, "reward_total_mean": 0.753753125667572, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9421913623809814, "rewards/meter/std": 0.0664798840880394, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.753753125667572, "rewards/total_composite/std": 0.0531839095056057, "sampling/importance_sampling_ratio/max": 1.4543837308883667, "sampling/importance_sampling_ratio/mean": 0.9989533424377441, "sampling/importance_sampling_ratio/min": 0.3693579435348511, "sampling/sampling_logp_difference/max": 0.9959890842437744, "sampling/sampling_logp_difference/mean": 0.014572631567716599, "step": 1349 }, { "clip_ratio/high_max": 0.007785359863191843, "clip_ratio/high_mean": 0.007785359863191843, "clip_ratio/low_mean": 0.015552184893749654, "clip_ratio/low_min": 0.015552184893749654, "clip_ratio/region_mean": 0.023337544756941497, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 63.625, "completions/mean_terminated_length": 63.625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.18225187622010708, "epoch": 0.054223400409687914, "frac_reward_zero_std": 0.0, "grad_norm": 4.666144847869873, "learning_rate": 5.912121212121212e-06, "loss": 0.0046, "num_tokens": 3046237.0, "reward": 0.5202029347419739, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5202029347419739, "reward_meter_std": 0.32961180806159973, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.32961177825927734, "reward_total_composite_mean": 0.5202029347419739, "reward_total_composite_std": 0.32961180806159973, "reward_total_mean": 0.5202029347419739, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5202029347419739, "rewards/meter/std": 0.32961180806159973, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5202029347419739, "rewards/total_composite/std": 0.32961180806159973, "sampling/importance_sampling_ratio/max": 1.43716299533844, "sampling/importance_sampling_ratio/mean": 0.9964258074760437, "sampling/importance_sampling_ratio/min": 0.2673254907131195, "sampling/sampling_logp_difference/max": 1.3192882537841797, "sampling/sampling_logp_difference/mean": 0.03747888654470444, "step": 1350 }, { "epoch": 0.054223400409687914, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 307.0, "eval_completions/max_terminated_length": 307.0, "eval_completions/mean_length": 172.09615384615384, "eval_completions/mean_terminated_length": 172.09615384615384, "eval_completions/min_length": 61.46153846153846, "eval_completions/min_terminated_length": 61.46153846153846, "eval_entropy": 0.14777196657199126, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 3046237.0, "eval_reward": 0.47231991015947783, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.8681816137754, "eval_reward_count_adherence_std": 0.14402351528406143, "eval_reward_meter_mean": 0.6744922651694372, "eval_reward_meter_std": 0.41629832753768337, "eval_reward_repeat_penalty_mean": 0.7968475085038406, "eval_reward_repeat_penalty_std": 0.16538272224939787, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.47231991015947783, "eval_reward_total_composite_std": 0.3407691912009166, "eval_reward_total_mean": 0.47231991015947783, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.8681816137754, "eval_rewards/count_adherence/std": 0.14402351528406143, "eval_rewards/meter/mean": 0.6744922651694372, "eval_rewards/meter/std": 0.41629832753768337, "eval_rewards/repeat_penalty/mean": 0.7968475085038406, "eval_rewards/repeat_penalty/std": 0.16538272224939787, "eval_rewards/total_composite/mean": 0.47231991015947783, "eval_rewards/total_composite/std": 0.3407691912009166, "eval_runtime": 59.1029, "eval_samples_per_second": 1.76, "eval_sampling/importance_sampling_ratio/max": 1.2998670798081617, "eval_sampling/importance_sampling_ratio/mean": 1.0026835478269136, "eval_sampling/importance_sampling_ratio/min": 0.37638011689369494, "eval_sampling/sampling_logp_difference/max": 1.005841695345365, "eval_sampling/sampling_logp_difference/mean": 0.016293376182707455, "eval_steps_per_second": 0.22, "step": 1350 }, { "clip_ratio/high_max": 0.03014001634437591, "clip_ratio/high_mean": 0.03014001634437591, "clip_ratio/low_mean": 0.008094369200989604, "clip_ratio/low_min": 0.008094369200989604, "clip_ratio/region_mean": 0.03823438554536551, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 134.0, "completions/mean_terminated_length": 134.0, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.21928934194147587, "epoch": 0.05426356589147287, "frac_reward_zero_std": 0.0, "grad_norm": 3.602323532104492, "learning_rate": 5.90909090909091e-06, "loss": 0.0176, "num_tokens": 3048749.0, "reward": 0.7975612878799438, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8950950503349304, "reward_meter_std": 0.18451449275016785, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.16415365040302277, "reward_total_composite_mean": 0.7975612878799438, "reward_total_composite_std": 0.16415363550186157, "reward_total_mean": 0.7975612878799438, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8950950503349304, "rewards/meter/std": 0.18451449275016785, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7975612878799438, "rewards/total_composite/std": 0.16415363550186157, "sampling/importance_sampling_ratio/max": 1.8117905855178833, "sampling/importance_sampling_ratio/mean": 1.0027098655700684, "sampling/importance_sampling_ratio/min": 0.1464216709136963, "sampling/sampling_logp_difference/max": 1.9212646484375, "sampling/sampling_logp_difference/mean": 0.036450449377298355, "step": 1351 }, { "clip_ratio/high_max": 0.011844757944345474, "clip_ratio/high_mean": 0.011844757944345474, "clip_ratio/low_mean": 0.011482007801532745, "clip_ratio/low_min": 0.011482007801532745, "clip_ratio/region_mean": 0.02332676574587822, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.25, "completions/mean_terminated_length": 32.25, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.14527534414082766, "epoch": 0.05430373137325782, "frac_reward_zero_std": 0.0, "grad_norm": 3.5144999027252197, "learning_rate": 5.906060606060607e-06, "loss": 0.0112, "num_tokens": 3050215.0, "reward": 0.9494646191596985, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9494646191596985, "reward_meter_std": 0.08898033201694489, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.08898033201694489, "reward_total_composite_mean": 0.9494646191596985, "reward_total_composite_std": 0.08898033201694489, "reward_total_mean": 0.9494646191596985, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9494646191596985, "rewards/meter/std": 0.08898033201694489, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9494646191596985, "rewards/total_composite/std": 0.08898033201694489, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9998236298561096, "sampling/importance_sampling_ratio/min": 0.35530397295951843, "sampling/sampling_logp_difference/max": 1.034781575202942, "sampling/sampling_logp_difference/mean": 0.03137016296386719, "step": 1352 }, { "clip_ratio/high_max": 0.031015241518616676, "clip_ratio/high_mean": 0.031015241518616676, "clip_ratio/low_mean": 0.014525586506351829, "clip_ratio/low_min": 0.014525586506351829, "clip_ratio/region_mean": 0.045540828024968505, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 68.25, "completions/mean_terminated_length": 68.25, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.19455979391932487, "epoch": 0.054343896855042775, "frac_reward_zero_std": 0.0, "grad_norm": 7.210882186889648, "learning_rate": 5.903030303030304e-06, "loss": 0.0173, "num_tokens": 3052065.0, "reward": 0.9697257876396179, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9697257876396179, "reward_meter_std": 0.039108145982027054, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03910814970731735, "reward_total_composite_mean": 0.9697257876396179, "reward_total_composite_std": 0.039108145982027054, "reward_total_mean": 0.9697257876396179, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9697257876396179, "rewards/meter/std": 0.039108145982027054, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9697257876396179, "rewards/total_composite/std": 0.039108145982027054, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003574252128601, "sampling/importance_sampling_ratio/min": 0.173557311296463, "sampling/sampling_logp_difference/max": 1.7512474060058594, "sampling/sampling_logp_difference/mean": 0.03900406137108803, "step": 1353 }, { "clip_ratio/high_max": 0.009631440974771976, "clip_ratio/high_mean": 0.009631440974771976, "clip_ratio/low_mean": 0.004853119375184178, "clip_ratio/low_min": 0.004853119375184178, "clip_ratio/region_mean": 0.014484560349956155, "completions/clipped_ratio": 0.0, "completions/max_length": 131.0, "completions/max_terminated_length": 131.0, "completions/mean_length": 128.875, "completions/mean_terminated_length": 128.875, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.07222465332597494, "epoch": 0.05438406233682773, "frac_reward_zero_std": 0.0, "grad_norm": 2.7349421977996826, "learning_rate": 5.9e-06, "loss": 0.0013, "num_tokens": 3054440.0, "reward": 0.7450422048568726, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9711472392082214, "reward_meter_std": 0.022641118615865707, "reward_repeat_penalty_mean": 0.7678571343421936, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.06550093740224838, "reward_total_composite_mean": 0.7450422048568726, "reward_total_composite_std": 0.06550093740224838, "reward_total_mean": 0.7450422048568726, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9711472392082214, "rewards/meter/std": 0.022641118615865707, "rewards/repeat_penalty/mean": 0.7678571343421936, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.7450422048568726, "rewards/total_composite/std": 0.06550093740224838, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016677379608154, "sampling/importance_sampling_ratio/min": 0.14024357497692108, "sampling/sampling_logp_difference/max": 1.9643745422363281, "sampling/sampling_logp_difference/mean": 0.01926073245704174, "step": 1354 }, { "clip_ratio/high_max": 0.04795599193312228, "clip_ratio/high_mean": 0.04795599193312228, "clip_ratio/low_mean": 0.011488970601931214, "clip_ratio/low_min": 0.011488970601931214, "clip_ratio/region_mean": 0.05944496253505349, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 33.75, "completions/mean_terminated_length": 33.75, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.3209527414292097, "epoch": 0.05442422781861268, "frac_reward_zero_std": 0.0, "grad_norm": 13.958351135253906, "learning_rate": 5.8969696969696975e-06, "loss": 0.017, "num_tokens": 3055950.0, "reward": 0.9982504844665527, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982504844665527, "reward_meter_std": 0.0002638747973833233, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00026387875550426543, "reward_total_composite_mean": 0.9982504844665527, "reward_total_composite_std": 0.0002638747973833233, "reward_total_mean": 0.9982504844665527, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982504844665527, "rewards/meter/std": 0.0002638747973833233, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982504844665527, "rewards/total_composite/std": 0.0002638747973833233, "sampling/importance_sampling_ratio/max": 1.8974801301956177, "sampling/importance_sampling_ratio/mean": 1.005803108215332, "sampling/importance_sampling_ratio/min": 0.19215981662273407, "sampling/sampling_logp_difference/max": 1.649427890777588, "sampling/sampling_logp_difference/mean": 0.0569852739572525, "step": 1355 }, { "clip_ratio/high_max": 0.007938507944345474, "clip_ratio/high_mean": 0.007938507944345474, "clip_ratio/low_mean": 0.023815523833036423, "clip_ratio/low_min": 0.023815523833036423, "clip_ratio/region_mean": 0.0317540317773819, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 31.625, "completions/mean_terminated_length": 31.625, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.0775892292149365, "epoch": 0.05446439330039764, "frac_reward_zero_std": 0.0, "grad_norm": 8.597617149353027, "learning_rate": 5.893939393939394e-06, "loss": 0.0063, "num_tokens": 3057379.0, "reward": 0.9754250049591064, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9754250049591064, "reward_meter_std": 0.010440711863338947, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010440702550113201, "reward_total_composite_mean": 0.9754250049591064, "reward_total_composite_std": 0.010440711863338947, "reward_total_mean": 0.9754250049591064, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9754250049591064, "rewards/meter/std": 0.010440711863338947, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9754250049591064, "rewards/total_composite/std": 0.010440711863338947, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9994874596595764, "sampling/importance_sampling_ratio/min": 0.2705341875553131, "sampling/sampling_logp_difference/max": 1.307356834411621, "sampling/sampling_logp_difference/mean": 0.02750094048678875, "step": 1356 }, { "clip_ratio/high_max": 0.028264103457331657, "clip_ratio/high_mean": 0.028264103457331657, "clip_ratio/low_mean": 0.01653005462139845, "clip_ratio/low_min": 0.01653005462139845, "clip_ratio/region_mean": 0.044794158078730106, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.375, "completions/mean_terminated_length": 61.375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2319873459637165, "epoch": 0.05450455878218259, "frac_reward_zero_std": 0.0, "grad_norm": 4.431914329528809, "learning_rate": 5.890909090909091e-06, "loss": -0.0049, "num_tokens": 3059262.0, "reward": 0.3558826744556427, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3558826744556427, "reward_meter_std": 0.09776563197374344, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09776563197374344, "reward_total_composite_mean": 0.3558826744556427, "reward_total_composite_std": 0.09776563197374344, "reward_total_mean": 0.3558826744556427, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3558826744556427, "rewards/meter/std": 0.09776563197374344, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3558826744556427, "rewards/total_composite/std": 0.09776563197374344, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008901596069336, "sampling/importance_sampling_ratio/min": 0.20266073942184448, "sampling/sampling_logp_difference/max": 1.596221923828125, "sampling/sampling_logp_difference/mean": 0.04409278184175491, "step": 1357 }, { "clip_ratio/high_max": 0.040843427646905184, "clip_ratio/high_mean": 0.040843427646905184, "clip_ratio/low_mean": 0.0016891892300918698, "clip_ratio/low_min": 0.0016891892300918698, "clip_ratio/region_mean": 0.042532616876997054, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.375, "completions/mean_terminated_length": 76.375, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.24777152389287949, "epoch": 0.054544724263967545, "frac_reward_zero_std": 0.0, "grad_norm": 6.204784393310547, "learning_rate": 5.887878787878788e-06, "loss": -0.003, "num_tokens": 3061057.0, "reward": 0.9842313528060913, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9842313528060913, "reward_meter_std": 0.035786159336566925, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03578615561127663, "reward_total_composite_mean": 0.9842313528060913, "reward_total_composite_std": 0.035786159336566925, "reward_total_mean": 0.9842313528060913, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9842313528060913, "rewards/meter/std": 0.035786159336566925, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9842313528060913, "rewards/total_composite/std": 0.035786159336566925, "sampling/importance_sampling_ratio/max": 1.8513555526733398, "sampling/importance_sampling_ratio/mean": 1.0056933164596558, "sampling/importance_sampling_ratio/min": 0.3075996935367584, "sampling/sampling_logp_difference/max": 1.1789560317993164, "sampling/sampling_logp_difference/mean": 0.042138196527957916, "step": 1358 }, { "clip_ratio/high_max": 0.026044346392154694, "clip_ratio/high_mean": 0.026044346392154694, "clip_ratio/low_mean": 0.006849315017461777, "clip_ratio/low_min": 0.006849315017461777, "clip_ratio/region_mean": 0.03289366140961647, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.125, "completions/mean_terminated_length": 72.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.31099611707031727, "epoch": 0.0545848897457525, "frac_reward_zero_std": 0.0, "grad_norm": 6.726247787475586, "learning_rate": 5.884848484848486e-06, "loss": 0.0055, "num_tokens": 3062858.0, "reward": 0.938990592956543, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.938990592956543, "reward_meter_std": 0.15821680426597595, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.15821681916713715, "reward_total_composite_mean": 0.938990592956543, "reward_total_composite_std": 0.15821680426597595, "reward_total_mean": 0.938990592956543, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.938990592956543, "rewards/meter/std": 0.15821680426597595, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.938990592956543, "rewards/total_composite/std": 0.15821680426597595, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.00753915309906, "sampling/importance_sampling_ratio/min": 0.19506093859672546, "sampling/sampling_logp_difference/max": 1.6344432830810547, "sampling/sampling_logp_difference/mean": 0.04102391377091408, "step": 1359 }, { "clip_ratio/high_max": 0.06222078762948513, "clip_ratio/high_mean": 0.06222078762948513, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/region_mean": 0.06916523212566972, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 69.75, "completions/mean_terminated_length": 69.75, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.3794647231698036, "epoch": 0.05462505522753745, "frac_reward_zero_std": 0.0, "grad_norm": 6.566722869873047, "learning_rate": 5.881818181818182e-06, "loss": 0.0188, "num_tokens": 3064544.0, "reward": 0.9943356513977051, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943356513977051, "reward_meter_std": 0.011215281672775745, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011215298436582088, "reward_total_composite_mean": 0.9943356513977051, "reward_total_composite_std": 0.011215281672775745, "reward_total_mean": 0.9943356513977051, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943356513977051, "rewards/meter/std": 0.011215281672775745, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943356513977051, "rewards/total_composite/std": 0.011215281672775745, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9980753064155579, "sampling/importance_sampling_ratio/min": 0.2655761241912842, "sampling/sampling_logp_difference/max": 1.3258538246154785, "sampling/sampling_logp_difference/mean": 0.07216734439134598, "step": 1360 }, { "clip_ratio/high_max": 0.051507277181372046, "clip_ratio/high_mean": 0.051507277181372046, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/region_mean": 0.057757277274504304, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 39.125, "completions/mean_terminated_length": 39.125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.40571761690080166, "epoch": 0.05466522070932241, "frac_reward_zero_std": 0.0, "grad_norm": 10.907339096069336, "learning_rate": 5.878787878787879e-06, "loss": 0.0262, "num_tokens": 3066089.0, "reward": 0.8477911949157715, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8477911949157715, "reward_meter_std": 0.32613706588745117, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3261370360851288, "reward_total_composite_mean": 0.8477911949157715, "reward_total_composite_std": 0.32613706588745117, "reward_total_mean": 0.8477911949157715, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8477911949157715, "rewards/meter/std": 0.32613706588745117, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8477911949157715, "rewards/total_composite/std": 0.32613706588745117, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0094305276870728, "sampling/importance_sampling_ratio/min": 0.15061601996421814, "sampling/sampling_logp_difference/max": 1.893021583557129, "sampling/sampling_logp_difference/mean": 0.07346105575561523, "step": 1361 }, { "clip_ratio/high_max": 0.06273440551012754, "clip_ratio/high_mean": 0.06273440551012754, "clip_ratio/low_mean": 0.015579710714519024, "clip_ratio/low_min": 0.015579710714519024, "clip_ratio/region_mean": 0.07831411622464657, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 72.875, "completions/mean_terminated_length": 72.875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.3851870410144329, "epoch": 0.05470538619110736, "frac_reward_zero_std": 0.0, "grad_norm": 5.575023651123047, "learning_rate": 5.875757575757576e-06, "loss": 0.0211, "num_tokens": 3068064.0, "reward": 0.8141105771064758, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8556216955184937, "reward_meter_std": 0.27104073762893677, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.27185213565826416, "reward_total_composite_mean": 0.8141105771064758, "reward_total_composite_std": 0.27185216546058655, "reward_total_mean": 0.8141105771064758, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8556216955184937, "rewards/meter/std": 0.27104073762893677, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.8141105771064758, "rewards/total_composite/std": 0.27185216546058655, "sampling/importance_sampling_ratio/max": 1.9909788370132446, "sampling/importance_sampling_ratio/mean": 0.9950591921806335, "sampling/importance_sampling_ratio/min": 0.15158410370349884, "sampling/sampling_logp_difference/max": 1.8866146802902222, "sampling/sampling_logp_difference/mean": 0.07884806394577026, "step": 1362 }, { "clip_ratio/high_max": 0.03211413393728435, "clip_ratio/high_mean": 0.03211413393728435, "clip_ratio/low_mean": 0.010135134682059288, "clip_ratio/low_min": 0.010135134682059288, "clip_ratio/region_mean": 0.04224926861934364, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.2922775577753782, "epoch": 0.054745551672892315, "frac_reward_zero_std": 0.0, "grad_norm": 14.677414894104004, "learning_rate": 5.872727272727273e-06, "loss": 0.0432, "num_tokens": 3069688.0, "reward": 0.601068377494812, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.601068377494812, "reward_meter_std": 0.3853451907634735, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3853451907634735, "reward_total_composite_mean": 0.601068377494812, "reward_total_composite_std": 0.3853451907634735, "reward_total_mean": 0.601068377494812, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.601068377494812, "rewards/meter/std": 0.3853451907634735, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.601068377494812, "rewards/total_composite/std": 0.3853451907634735, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0056836605072021, "sampling/importance_sampling_ratio/min": 0.14872407913208008, "sampling/sampling_logp_difference/max": 1.9056625366210938, "sampling/sampling_logp_difference/mean": 0.05407094210386276, "step": 1363 }, { "clip_ratio/high_max": 0.010213951813057065, "clip_ratio/high_mean": 0.010213951813057065, "clip_ratio/low_mean": 0.004132513655349612, "clip_ratio/low_min": 0.004132513655349612, "clip_ratio/region_mean": 0.014346465468406677, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 60.875, "completions/mean_terminated_length": 60.875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.10629169084131718, "epoch": 0.05478571715467727, "frac_reward_zero_std": 0.0, "grad_norm": 3.343250274658203, "learning_rate": 5.8696969696969694e-06, "loss": -0.0013, "num_tokens": 3071527.0, "reward": 0.987983763217926, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.987983763217926, "reward_meter_std": 0.010048545897006989, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010048558004200459, "reward_total_composite_mean": 0.987983763217926, "reward_total_composite_std": 0.010048545897006989, "reward_total_mean": 0.987983763217926, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.987983763217926, "rewards/meter/std": 0.010048545897006989, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.987983763217926, "rewards/total_composite/std": 0.010048545897006989, "sampling/importance_sampling_ratio/max": 1.870787501335144, "sampling/importance_sampling_ratio/mean": 0.9991443157196045, "sampling/importance_sampling_ratio/min": 0.41118231415748596, "sampling/sampling_logp_difference/max": 0.8887186050415039, "sampling/sampling_logp_difference/mean": 0.020703302696347237, "step": 1364 }, { "clip_ratio/high_max": 0.01694969367235899, "clip_ratio/high_mean": 0.01694969367235899, "clip_ratio/low_mean": 0.032539316453039646, "clip_ratio/low_min": 0.032539316453039646, "clip_ratio/region_mean": 0.049489010125398636, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 91.25, "completions/mean_terminated_length": 91.25, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.3218696843832731, "epoch": 0.05482588263646222, "frac_reward_zero_std": 0.0, "grad_norm": 5.388495445251465, "learning_rate": 5.8666666666666675e-06, "loss": 0.0227, "num_tokens": 3073505.0, "reward": 0.09825273603200912, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.09923887252807617, "reward_meter_std": 0.17276984453201294, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.17330722510814667, "reward_total_composite_mean": 0.09825273603200912, "reward_total_composite_std": 0.17330721020698547, "reward_total_mean": 0.09825273603200912, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.09923887252807617, "rewards/meter/std": 0.17276984453201294, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.09825273603200912, "rewards/total_composite/std": 0.17330721020698547, "sampling/importance_sampling_ratio/max": 1.888500452041626, "sampling/importance_sampling_ratio/mean": 1.003082036972046, "sampling/importance_sampling_ratio/min": 0.2242598831653595, "sampling/sampling_logp_difference/max": 1.4949496984481812, "sampling/sampling_logp_difference/mean": 0.05340426787734032, "step": 1365 }, { "clip_ratio/high_max": 0.055123385740444064, "clip_ratio/high_mean": 0.055123385740444064, "clip_ratio/low_mean": 0.019174665678292513, "clip_ratio/low_min": 0.019174665678292513, "clip_ratio/region_mean": 0.07429805141873658, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 74.5, "completions/mean_terminated_length": 74.5, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.37352898344397545, "epoch": 0.05486604811824718, "frac_reward_zero_std": 0.0, "grad_norm": 8.658242225646973, "learning_rate": 5.863636363636364e-06, "loss": 0.0137, "num_tokens": 3075341.0, "reward": 0.9018104076385498, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9018104076385498, "reward_meter_std": 0.15291984379291534, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.15291984379291534, "reward_total_composite_mean": 0.9018104076385498, "reward_total_composite_std": 0.15291984379291534, "reward_total_mean": 0.9018104076385498, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9018104076385498, "rewards/meter/std": 0.15291984379291534, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9018104076385498, "rewards/total_composite/std": 0.15291984379291534, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0000320672988892, "sampling/importance_sampling_ratio/min": 0.1663859635591507, "sampling/sampling_logp_difference/max": 1.793445110321045, "sampling/sampling_logp_difference/mean": 0.08389697223901749, "step": 1366 }, { "clip_ratio/high_max": 0.014278159011155367, "clip_ratio/high_mean": 0.014278159011155367, "clip_ratio/low_mean": 0.013644688995555043, "clip_ratio/low_min": 0.013644688995555043, "clip_ratio/region_mean": 0.02792284800671041, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 62.625, "completions/mean_terminated_length": 62.625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.09856067411601543, "epoch": 0.05490621360003213, "frac_reward_zero_std": 0.0, "grad_norm": 7.315425395965576, "learning_rate": 5.860606060606061e-06, "loss": 0.0233, "num_tokens": 3076962.0, "reward": 0.9573122262954712, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9573122262954712, "reward_meter_std": 0.04094774276018143, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04094773158431053, "reward_total_composite_mean": 0.9573122262954712, "reward_total_composite_std": 0.04094774276018143, "reward_total_mean": 0.9573122262954712, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9573122262954712, "rewards/meter/std": 0.04094774276018143, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9573122262954712, "rewards/total_composite/std": 0.04094774276018143, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.994452714920044, "sampling/importance_sampling_ratio/min": 0.17685790359973907, "sampling/sampling_logp_difference/max": 1.7324086427688599, "sampling/sampling_logp_difference/mean": 0.03150878846645355, "step": 1367 }, { "clip_ratio/high_max": 0.03145654499530792, "clip_ratio/high_mean": 0.03145654499530792, "clip_ratio/low_mean": 0.028853498864918947, "clip_ratio/low_min": 0.028853498864918947, "clip_ratio/region_mean": 0.06031004386022687, "completions/clipped_ratio": 0.0, "completions/max_length": 218.0, "completions/max_terminated_length": 218.0, "completions/mean_length": 186.625, "completions/mean_terminated_length": 186.625, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.35142070427536964, "epoch": 0.054946379081817084, "frac_reward_zero_std": 0.0, "grad_norm": 3.8533852100372314, "learning_rate": 5.8575757575757584e-06, "loss": 0.0273, "num_tokens": 3079903.0, "reward": 0.644354522228241, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.07715168595314026, "reward_meter_mean": 0.7870528697967529, "reward_meter_std": 0.30166095495224, "reward_repeat_penalty_mean": 0.9494949579238892, "reward_repeat_penalty_std": 0.07303925603628159, "reward_std": 0.23647886514663696, "reward_total_composite_mean": 0.644354522228241, "reward_total_composite_std": 0.23647885024547577, "reward_total_mean": 0.644354522228241, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.07715168595314026, "rewards/meter/mean": 0.7870528697967529, "rewards/meter/std": 0.30166095495224, "rewards/repeat_penalty/mean": 0.9494949579238892, "rewards/repeat_penalty/std": 0.07303925603628159, "rewards/total_composite/mean": 0.644354522228241, "rewards/total_composite/std": 0.23647885024547577, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9927666783332825, "sampling/importance_sampling_ratio/min": 0.025069857016205788, "sampling/sampling_logp_difference/max": 3.686089038848877, "sampling/sampling_logp_difference/mean": 0.07801329344511032, "step": 1368 }, { "clip_ratio/high_max": 0.00881484942510724, "clip_ratio/high_mean": 0.00881484942510724, "clip_ratio/low_mean": 0.04172989120706916, "clip_ratio/low_min": 0.04172989120706916, "clip_ratio/region_mean": 0.0505447406321764, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 109.625, "completions/mean_terminated_length": 109.625, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.2641350254416466, "epoch": 0.05498654456360204, "frac_reward_zero_std": 0.0, "grad_norm": 6.314200401306152, "learning_rate": 5.854545454545455e-06, "loss": -0.0105, "num_tokens": 3082028.0, "reward": 0.09159435331821442, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.09159435331821442, "reward_meter_std": 0.1886763721704483, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1886763721704483, "reward_total_composite_mean": 0.09159435331821442, "reward_total_composite_std": 0.1886763721704483, "reward_total_mean": 0.09159435331821442, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.09159435331821442, "rewards/meter/std": 0.1886763721704483, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.09159435331821442, "rewards/total_composite/std": 0.1886763721704483, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008489966392517, "sampling/importance_sampling_ratio/min": 0.2004692256450653, "sampling/sampling_logp_difference/max": 1.6070945262908936, "sampling/sampling_logp_difference/mean": 0.056608088314533234, "step": 1369 }, { "clip_ratio/high_max": 0.017049296759068966, "clip_ratio/high_mean": 0.017049296759068966, "clip_ratio/low_mean": 0.013002454303205013, "clip_ratio/low_min": 0.013002454303205013, "clip_ratio/region_mean": 0.03005175106227398, "completions/clipped_ratio": 0.0, "completions/max_length": 162.0, "completions/max_terminated_length": 162.0, "completions/mean_length": 147.125, "completions/mean_terminated_length": 147.125, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.1116212410852313, "epoch": 0.05502671004538699, "frac_reward_zero_std": 0.0, "grad_norm": 2.687344551086426, "learning_rate": 5.851515151515152e-06, "loss": 0.0008, "num_tokens": 3084733.0, "reward": 0.4428256154060364, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.639522135257721, "reward_meter_std": 0.19242770969867706, "reward_repeat_penalty_mean": 0.7559523582458496, "reward_repeat_penalty_std": 0.17674170434474945, "reward_std": 0.15113258361816406, "reward_total_composite_mean": 0.4428256154060364, "reward_total_composite_std": 0.15113258361816406, "reward_total_mean": 0.4428256154060364, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.639522135257721, "rewards/meter/std": 0.19242770969867706, "rewards/repeat_penalty/mean": 0.7559523582458496, "rewards/repeat_penalty/std": 0.17674170434474945, "rewards/total_composite/mean": 0.4428256154060364, "rewards/total_composite/std": 0.15113258361816406, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9980568289756775, "sampling/importance_sampling_ratio/min": 0.040033698081970215, "sampling/sampling_logp_difference/max": 3.218033790588379, "sampling/sampling_logp_difference/mean": 0.030883800238370895, "step": 1370 }, { "clip_ratio/high_max": 0.022983014350757003, "clip_ratio/high_mean": 0.022983014350757003, "clip_ratio/low_mean": 0.010699023492634296, "clip_ratio/low_min": 0.010699023492634296, "clip_ratio/region_mean": 0.0336820378433913, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 125.375, "completions/mean_terminated_length": 125.375, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.18958070501685143, "epoch": 0.055066875527171946, "frac_reward_zero_std": 0.0, "grad_norm": 3.544004440307617, "learning_rate": 5.8484848484848485e-06, "loss": 0.021, "num_tokens": 3087080.0, "reward": 0.727212131023407, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7613728642463684, "reward_meter_std": 0.35931819677352905, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.34967654943466187, "reward_total_composite_mean": 0.727212131023407, "reward_total_composite_std": 0.34967657923698425, "reward_total_mean": 0.727212131023407, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7613728642463684, "rewards/meter/std": 0.35931819677352905, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.727212131023407, "rewards/total_composite/std": 0.34967657923698425, "sampling/importance_sampling_ratio/max": 1.806413173675537, "sampling/importance_sampling_ratio/mean": 1.009077787399292, "sampling/importance_sampling_ratio/min": 0.4156763255596161, "sampling/sampling_logp_difference/max": 0.8778483867645264, "sampling/sampling_logp_difference/mean": 0.034298431128263474, "step": 1371 }, { "clip_ratio/high_max": 0.03647972596809268, "clip_ratio/high_mean": 0.03647972596809268, "clip_ratio/low_mean": 0.009019391611218452, "clip_ratio/low_min": 0.009019391611218452, "clip_ratio/region_mean": 0.04549911757931113, "completions/clipped_ratio": 0.0, "completions/max_length": 159.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 153.625, "completions/mean_terminated_length": 153.625, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.26759735494852066, "epoch": 0.0551070410089569, "frac_reward_zero_std": 0.0, "grad_norm": 5.1783366203308105, "learning_rate": 5.845454545454547e-06, "loss": 0.0079, "num_tokens": 3089717.0, "reward": 0.05596386641263962, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.05596386641263962, "reward_meter_std": 0.034602247178554535, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03460225090384483, "reward_total_composite_mean": 0.05596386641263962, "reward_total_composite_std": 0.034602247178554535, "reward_total_mean": 0.05596386641263962, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.05596386641263962, "rewards/meter/std": 0.034602247178554535, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.05596386641263962, "rewards/total_composite/std": 0.034602247178554535, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9990571141242981, "sampling/importance_sampling_ratio/min": 0.004479795694351196, "sampling/sampling_logp_difference/max": 5.408177852630615, "sampling/sampling_logp_difference/mean": 0.07258274406194687, "step": 1372 }, { "clip_ratio/high_max": 0.01714563579298556, "clip_ratio/high_mean": 0.01714563579298556, "clip_ratio/low_mean": 0.0139405254740268, "clip_ratio/low_min": 0.0139405254740268, "clip_ratio/region_mean": 0.031086161267012358, "completions/clipped_ratio": 0.0, "completions/max_length": 272.0, "completions/max_terminated_length": 272.0, "completions/mean_length": 265.625, "completions/mean_terminated_length": 265.625, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.22261576727032661, "epoch": 0.055147206490741854, "frac_reward_zero_std": 0.0, "grad_norm": 2.5243654251098633, "learning_rate": 5.842424242424243e-06, "loss": 0.0156, "num_tokens": 3093370.0, "reward": 0.6546376943588257, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9450531005859375, "reward_meter_std": 0.09262127429246902, "reward_repeat_penalty_mean": 0.7980769872665405, "reward_repeat_penalty_std": 0.10019000619649887, "reward_std": 0.05606428161263466, "reward_total_composite_mean": 0.6546376943588257, "reward_total_composite_std": 0.05606427788734436, "reward_total_mean": 0.6546376943588257, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9450531005859375, "rewards/meter/std": 0.09262127429246902, "rewards/repeat_penalty/mean": 0.7980769872665405, "rewards/repeat_penalty/std": 0.10019000619649887, "rewards/total_composite/mean": 0.6546376943588257, "rewards/total_composite/std": 0.05606427788734436, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016841888427734, "sampling/importance_sampling_ratio/min": 0.043650176376104355, "sampling/sampling_logp_difference/max": 3.1315479278564453, "sampling/sampling_logp_difference/mean": 0.036455702036619186, "step": 1373 }, { "clip_ratio/high_max": 0.020197488833218813, "clip_ratio/high_mean": 0.020197488833218813, "clip_ratio/low_mean": 0.011315496172755957, "clip_ratio/low_min": 0.011315496172755957, "clip_ratio/region_mean": 0.03151298500597477, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.1000348855741322, "epoch": 0.05518737197252681, "frac_reward_zero_std": 0.0, "grad_norm": 5.411123752593994, "learning_rate": 5.83939393939394e-06, "loss": 0.0443, "num_tokens": 3095138.0, "reward": 0.9784355759620667, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9784355759620667, "reward_meter_std": 0.027008963748812675, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.027008960023522377, "reward_total_composite_mean": 0.9784355759620667, "reward_total_composite_std": 0.027008963748812675, "reward_total_mean": 0.9784355759620667, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9784355759620667, "rewards/meter/std": 0.027008963748812675, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9784355759620667, "rewards/total_composite/std": 0.027008963748812675, "sampling/importance_sampling_ratio/max": 1.9240130186080933, "sampling/importance_sampling_ratio/mean": 1.0000766515731812, "sampling/importance_sampling_ratio/min": 0.17678546905517578, "sampling/sampling_logp_difference/max": 1.732818365097046, "sampling/sampling_logp_difference/mean": 0.03236331045627594, "step": 1374 }, { "clip_ratio/high_max": 0.04311362677253783, "clip_ratio/high_mean": 0.04311362677253783, "clip_ratio/low_mean": 0.00937500037252903, "clip_ratio/low_min": 0.00937500037252903, "clip_ratio/region_mean": 0.05248862714506686, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 40.25, "completions/mean_terminated_length": 40.25, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.47422025352716446, "epoch": 0.05522753745431176, "frac_reward_zero_std": 0.0, "grad_norm": 10.806543350219727, "learning_rate": 5.836363636363637e-06, "loss": 0.0116, "num_tokens": 3096588.0, "reward": 0.9826058149337769, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9826058149337769, "reward_meter_std": 0.01861639879643917, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.018616393208503723, "reward_total_composite_mean": 0.9826058149337769, "reward_total_composite_std": 0.01861639879643917, "reward_total_mean": 0.9826058149337769, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9826058149337769, "rewards/meter/std": 0.01861639879643917, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9826058149337769, "rewards/total_composite/std": 0.01861639879643917, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0129060745239258, "sampling/importance_sampling_ratio/min": 0.25509995222091675, "sampling/sampling_logp_difference/max": 1.3660998344421387, "sampling/sampling_logp_difference/mean": 0.06725956499576569, "step": 1375 }, { "clip_ratio/high_max": 0.019607282243669033, "clip_ratio/high_mean": 0.019607282243669033, "clip_ratio/low_mean": 0.021116425283253193, "clip_ratio/low_min": 0.021116425283253193, "clip_ratio/region_mean": 0.040723707526922226, "completions/clipped_ratio": 0.0, "completions/max_length": 256.0, "completions/max_terminated_length": 256.0, "completions/mean_length": 245.625, "completions/mean_terminated_length": 245.625, "completions/min_length": 241.0, "completions/min_terminated_length": 241.0, "entropy": 0.2056045476347208, "epoch": 0.055267702936096716, "frac_reward_zero_std": 0.0, "grad_norm": 4.0455732345581055, "learning_rate": 5.833333333333334e-06, "loss": 0.026, "num_tokens": 3100217.0, "reward": 0.4196595549583435, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.735993504524231, "reward_meter_std": 0.3604574203491211, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10079052299261093, "reward_std": 0.19926169514656067, "reward_total_composite_mean": 0.4196595549583435, "reward_total_composite_std": 0.19926168024539948, "reward_total_mean": 0.4196595549583435, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.735993504524231, "rewards/meter/std": 0.3604574203491211, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10079052299261093, "rewards/total_composite/mean": 0.4196595549583435, "rewards/total_composite/std": 0.19926168024539948, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9991115927696228, "sampling/importance_sampling_ratio/min": 0.00024855564697645605, "sampling/sampling_logp_difference/max": 8.299843788146973, "sampling/sampling_logp_difference/mean": 0.06053018942475319, "step": 1376 }, { "clip_ratio/high_max": 0.028373429435305297, "clip_ratio/high_mean": 0.028373429435305297, "clip_ratio/low_mean": 0.02467968501150608, "clip_ratio/low_min": 0.02467968501150608, "clip_ratio/region_mean": 0.05305311444681138, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 79.5, "completions/mean_terminated_length": 79.5, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.3501878045499325, "epoch": 0.05530786841788167, "frac_reward_zero_std": 0.0, "grad_norm": 5.782622814178467, "learning_rate": 5.83030303030303e-06, "loss": 0.0091, "num_tokens": 3102173.0, "reward": 0.9965003728866577, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9965003728866577, "reward_meter_std": 0.0018758656224235892, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0018758544465526938, "reward_total_composite_mean": 0.9965003728866577, "reward_total_composite_std": 0.0018758656224235892, "reward_total_mean": 0.9965003728866577, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9965003728866577, "rewards/meter/std": 0.0018758656224235892, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9965003728866577, "rewards/total_composite/std": 0.0018758656224235892, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003879189491272, "sampling/importance_sampling_ratio/min": 0.11376892775297165, "sampling/sampling_logp_difference/max": 2.173585891723633, "sampling/sampling_logp_difference/mean": 0.057900648564100266, "step": 1377 }, { "clip_ratio/high_max": 0.022769354982301593, "clip_ratio/high_mean": 0.022769354982301593, "clip_ratio/low_mean": 0.010481631616130471, "clip_ratio/low_min": 0.010481631616130471, "clip_ratio/region_mean": 0.033250986598432064, "completions/clipped_ratio": 0.0, "completions/max_length": 189.0, "completions/max_terminated_length": 189.0, "completions/mean_length": 172.5, "completions/mean_terminated_length": 172.5, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.2811085060238838, "epoch": 0.055348033899666624, "frac_reward_zero_std": 0.0, "grad_norm": 6.239497184753418, "learning_rate": 5.8272727272727285e-06, "loss": -0.0018, "num_tokens": 3104993.0, "reward": 0.8381317853927612, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.9290133714675903, "reward_meter_std": 0.17080947756767273, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.1523481160402298, "reward_total_composite_mean": 0.8381317853927612, "reward_total_composite_std": 0.152348130941391, "reward_total_mean": 0.8381317853927612, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.9290133714675903, "rewards/meter/std": 0.17080947756767273, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.8381317853927612, "rewards/total_composite/std": 0.152348130941391, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0066179037094116, "sampling/importance_sampling_ratio/min": 0.07610113173723221, "sampling/sampling_logp_difference/max": 2.5756921768188477, "sampling/sampling_logp_difference/mean": 0.04662080481648445, "step": 1378 }, { "clip_ratio/high_max": 0.006693349685519934, "clip_ratio/high_mean": 0.006693349685519934, "clip_ratio/low_mean": 0.007781740743666887, "clip_ratio/low_min": 0.007781740743666887, "clip_ratio/region_mean": 0.014475090429186821, "completions/clipped_ratio": 0.0, "completions/max_length": 248.0, "completions/max_terminated_length": 248.0, "completions/mean_length": 241.75, "completions/mean_terminated_length": 241.75, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.09616232290863991, "epoch": 0.05538819938145158, "frac_reward_zero_std": 0.0, "grad_norm": 2.3529670238494873, "learning_rate": 5.824242424242425e-06, "loss": -0.0035, "num_tokens": 3108711.0, "reward": 0.4973941147327423, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9438363313674927, "reward_meter_std": 0.13904310762882233, "reward_repeat_penalty_mean": 0.7250000238418579, "reward_repeat_penalty_std": 0.115125872194767, "reward_std": 0.1098327562212944, "reward_total_composite_mean": 0.4973941147327423, "reward_total_composite_std": 0.1098327562212944, "reward_total_mean": 0.4973941147327423, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9438363313674927, "rewards/meter/std": 0.13904310762882233, "rewards/repeat_penalty/mean": 0.7250000238418579, "rewards/repeat_penalty/std": 0.115125872194767, "rewards/total_composite/mean": 0.4973941147327423, "rewards/total_composite/std": 0.1098327562212944, "sampling/importance_sampling_ratio/max": 1.9625972509384155, "sampling/importance_sampling_ratio/mean": 1.0025800466537476, "sampling/importance_sampling_ratio/min": 0.17987366020679474, "sampling/sampling_logp_difference/max": 1.7155005931854248, "sampling/sampling_logp_difference/mean": 0.019148869439959526, "step": 1379 }, { "clip_ratio/high_max": 0.008928571827709675, "clip_ratio/high_mean": 0.008928571827709675, "clip_ratio/low_mean": 0.01270932296756655, "clip_ratio/low_min": 0.01270932296756655, "clip_ratio/region_mean": 0.021637894795276225, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.1250211326405406, "epoch": 0.05542836486323653, "frac_reward_zero_std": 0.0, "grad_norm": 3.860546112060547, "learning_rate": 5.821212121212122e-06, "loss": 0.0058, "num_tokens": 3110495.0, "reward": 0.014178031124174595, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.014178031124174595, "reward_meter_std": 0.004543017130345106, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004543016664683819, "reward_total_composite_mean": 0.014178031124174595, "reward_total_composite_std": 0.004543017130345106, "reward_total_mean": 0.014178031124174595, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.014178031124174595, "rewards/meter/std": 0.004543017130345106, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.014178031124174595, "rewards/total_composite/std": 0.004543017130345106, "sampling/importance_sampling_ratio/max": 1.7626913785934448, "sampling/importance_sampling_ratio/mean": 0.9993897676467896, "sampling/importance_sampling_ratio/min": 0.051215339452028275, "sampling/sampling_logp_difference/max": 2.9717161655426025, "sampling/sampling_logp_difference/mean": 0.02862457185983658, "step": 1380 }, { "clip_ratio/high_max": 0.003260869416408241, "clip_ratio/high_mean": 0.003260869416408241, "clip_ratio/low_mean": 0.003251499147154391, "clip_ratio/low_min": 0.003251499147154391, "clip_ratio/region_mean": 0.006512368563562632, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 115.25, "completions/mean_terminated_length": 115.25, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.039332801941782236, "epoch": 0.055468530345021486, "frac_reward_zero_std": 0.0, "grad_norm": 1.1249967813491821, "learning_rate": 5.8181818181818185e-06, "loss": 0.0024, "num_tokens": 3112713.0, "reward": 0.319552481174469, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9941756725311279, "reward_meter_std": 0.00028240663232281804, "reward_repeat_penalty_mean": 0.3214285969734192, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06571788340806961, "reward_total_composite_mean": 0.319552481174469, "reward_total_composite_std": 0.0657178983092308, "reward_total_mean": 0.319552481174469, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9941756725311279, "rewards/meter/std": 0.00028240663232281804, "rewards/repeat_penalty/mean": 0.3214285969734192, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.319552481174469, "rewards/total_composite/std": 0.0657178983092308, "sampling/importance_sampling_ratio/max": 1.2996612787246704, "sampling/importance_sampling_ratio/mean": 0.9989582896232605, "sampling/importance_sampling_ratio/min": 0.3473857343196869, "sampling/sampling_logp_difference/max": 1.0573194026947021, "sampling/sampling_logp_difference/mean": 0.009904337115585804, "step": 1381 }, { "clip_ratio/high_max": 0.017227754462510347, "clip_ratio/high_mean": 0.017227754462510347, "clip_ratio/low_mean": 0.0058121298789046705, "clip_ratio/low_min": 0.0058121298789046705, "clip_ratio/region_mean": 0.023039884341415018, "completions/clipped_ratio": 0.0, "completions/max_length": 155.0, "completions/max_terminated_length": 155.0, "completions/mean_length": 151.875, "completions/mean_terminated_length": 151.875, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.1818629987537861, "epoch": 0.05550869582680644, "frac_reward_zero_std": 0.0, "grad_norm": 4.528812408447266, "learning_rate": 5.815151515151516e-06, "loss": 0.0069, "num_tokens": 3115264.0, "reward": 0.7798627614974976, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9926921725273132, "reward_meter_std": 0.008148393593728542, "reward_repeat_penalty_mean": 0.7857142686843872, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.07480985671281815, "reward_total_composite_mean": 0.7798627614974976, "reward_total_composite_std": 0.07480987906455994, "reward_total_mean": 0.7798627614974976, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9926921725273132, "rewards/meter/std": 0.008148393593728542, "rewards/repeat_penalty/mean": 0.7857142686843872, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.7798627614974976, "rewards/total_composite/std": 0.07480987906455994, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0049320459365845, "sampling/importance_sampling_ratio/min": 0.29014089703559875, "sampling/sampling_logp_difference/max": 1.2373886108398438, "sampling/sampling_logp_difference/mean": 0.030164044350385666, "step": 1382 }, { "clip_ratio/high_max": 0.02543990476988256, "clip_ratio/high_mean": 0.02543990476988256, "clip_ratio/low_mean": 0.001168224262073636, "clip_ratio/low_min": 0.001168224262073636, "clip_ratio/region_mean": 0.026608129031956196, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 103.75, "completions/mean_terminated_length": 103.75, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.18817367777228355, "epoch": 0.055548861308591393, "frac_reward_zero_std": 0.0, "grad_norm": 3.192974805831909, "learning_rate": 5.812121212121212e-06, "loss": 0.0148, "num_tokens": 3117462.0, "reward": 0.9773154258728027, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9773154258728027, "reward_meter_std": 0.0542588047683239, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0542587973177433, "reward_total_composite_mean": 0.9773154258728027, "reward_total_composite_std": 0.0542588047683239, "reward_total_mean": 0.9773154258728027, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9773154258728027, "rewards/meter/std": 0.0542588047683239, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9773154258728027, "rewards/total_composite/std": 0.0542588047683239, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0025712251663208, "sampling/importance_sampling_ratio/min": 0.331994891166687, "sampling/sampling_logp_difference/max": 1.3268545866012573, "sampling/sampling_logp_difference/mean": 0.03143962472677231, "step": 1383 }, { "clip_ratio/high_max": 0.01638579391874373, "clip_ratio/high_mean": 0.01638579391874373, "clip_ratio/low_mean": 0.014534936985000968, "clip_ratio/low_min": 0.014534936985000968, "clip_ratio/region_mean": 0.030920730903744698, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.375, "completions/mean_terminated_length": 69.375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.13980651926249266, "epoch": 0.05558902679037635, "frac_reward_zero_std": 0.0, "grad_norm": 7.283441543579102, "learning_rate": 5.8090909090909095e-06, "loss": 0.0122, "num_tokens": 3119369.0, "reward": 0.99678635597229, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99678635597229, "reward_meter_std": 0.001158631988801062, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001158619881607592, "reward_total_composite_mean": 0.99678635597229, "reward_total_composite_std": 0.001158631988801062, "reward_total_mean": 0.99678635597229, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99678635597229, "rewards/meter/std": 0.001158631988801062, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99678635597229, "rewards/total_composite/std": 0.001158631988801062, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015077590942383, "sampling/importance_sampling_ratio/min": 0.009025109931826591, "sampling/sampling_logp_difference/max": 4.707744598388672, "sampling/sampling_logp_difference/mean": 0.042962342500686646, "step": 1384 }, { "clip_ratio/high_max": 0.00390625, "clip_ratio/high_mean": 0.00390625, "clip_ratio/low_mean": 0.007464349502697587, "clip_ratio/low_min": 0.007464349502697587, "clip_ratio/region_mean": 0.011370599502697587, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 32.25, "completions/mean_terminated_length": 32.25, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.1377080613747239, "epoch": 0.0556291922721613, "frac_reward_zero_std": 0.0, "grad_norm": 8.730010032653809, "learning_rate": 5.806060606060606e-06, "loss": 0.0196, "num_tokens": 3120803.0, "reward": 0.9718137979507446, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9718137979507446, "reward_meter_std": 0.05166900157928467, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.051669009029865265, "reward_total_composite_mean": 0.9718137979507446, "reward_total_composite_std": 0.05166900157928467, "reward_total_mean": 0.9718137979507446, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9718137979507446, "rewards/meter/std": 0.05166900157928467, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9718137979507446, "rewards/total_composite/std": 0.05166900157928467, "sampling/importance_sampling_ratio/max": 1.604980230331421, "sampling/importance_sampling_ratio/mean": 1.0018343925476074, "sampling/importance_sampling_ratio/min": 0.5025351643562317, "sampling/sampling_logp_difference/max": 0.6880896091461182, "sampling/sampling_logp_difference/mean": 0.01950211450457573, "step": 1385 }, { "clip_ratio/high_max": 0.010490705259144306, "clip_ratio/high_mean": 0.010490705259144306, "clip_ratio/low_mean": 0.017045454820618033, "clip_ratio/low_min": 0.017045454820618033, "clip_ratio/region_mean": 0.02753616007976234, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 58.5, "completions/mean_terminated_length": 58.5, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.29168289341032505, "epoch": 0.055669357753946255, "frac_reward_zero_std": 0.0, "grad_norm": 5.677158832550049, "learning_rate": 5.803030303030304e-06, "loss": -0.0023, "num_tokens": 3122583.0, "reward": 0.307847261428833, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.307847261428833, "reward_meter_std": 0.2652658224105835, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2652658224105835, "reward_total_composite_mean": 0.307847261428833, "reward_total_composite_std": 0.2652658224105835, "reward_total_mean": 0.307847261428833, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.307847261428833, "rewards/meter/std": 0.2652658224105835, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.307847261428833, "rewards/total_composite/std": 0.2652658224105835, "sampling/importance_sampling_ratio/max": 1.7868449687957764, "sampling/importance_sampling_ratio/mean": 1.0097620487213135, "sampling/importance_sampling_ratio/min": 0.35695940256118774, "sampling/sampling_logp_difference/max": 1.0301332473754883, "sampling/sampling_logp_difference/mean": 0.04078377038240433, "step": 1386 }, { "clip_ratio/high_max": 0.01465759938582778, "clip_ratio/high_mean": 0.01465759938582778, "clip_ratio/low_mean": 0.011592080816626549, "clip_ratio/low_min": 0.011592080816626549, "clip_ratio/region_mean": 0.02624968020245433, "completions/clipped_ratio": 0.0, "completions/max_length": 235.0, "completions/max_terminated_length": 235.0, "completions/mean_length": 223.125, "completions/mean_terminated_length": 223.125, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.16380778048187494, "epoch": 0.05570952323573121, "frac_reward_zero_std": 0.0, "grad_norm": 3.5950727462768555, "learning_rate": 5.8e-06, "loss": 0.0117, "num_tokens": 3125864.0, "reward": 0.5415959358215332, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8055555820465088, "reward_count_adherence_std": 0.05143444985151291, "reward_meter_mean": 0.9973956346511841, "reward_meter_std": 0.0017723083728924394, "reward_repeat_penalty_mean": 0.6742216348648071, "reward_repeat_penalty_std": 0.1199769675731659, "reward_std": 0.10306404531002045, "reward_total_composite_mean": 0.5415959358215332, "reward_total_composite_std": 0.10306404531002045, "reward_total_mean": 0.5415959358215332, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8055555820465088, "rewards/count_adherence/std": 0.05143444985151291, "rewards/meter/mean": 0.9973956346511841, "rewards/meter/std": 0.0017723083728924394, "rewards/repeat_penalty/mean": 0.6742216348648071, "rewards/repeat_penalty/std": 0.1199769675731659, "rewards/total_composite/mean": 0.5415959358215332, "rewards/total_composite/std": 0.10306404531002045, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015687942504883, "sampling/importance_sampling_ratio/min": 1.7721692557870483e-08, "sampling/sampling_logp_difference/max": 17.84847640991211, "sampling/sampling_logp_difference/mean": 0.058670658618211746, "step": 1387 }, { "clip_ratio/high_max": 0.009384893695823848, "clip_ratio/high_mean": 0.009384893695823848, "clip_ratio/low_mean": 0.007462686393409967, "clip_ratio/low_min": 0.007462686393409967, "clip_ratio/region_mean": 0.016847580089233816, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.12053523678332567, "epoch": 0.05574968871751616, "frac_reward_zero_std": 0.0, "grad_norm": 3.6467227935791016, "learning_rate": 5.796969696969698e-06, "loss": 0.0122, "num_tokens": 3127767.0, "reward": 0.9714630842208862, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9714630842208862, "reward_meter_std": 0.009149931371212006, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009149930439889431, "reward_total_composite_mean": 0.9714630842208862, "reward_total_composite_std": 0.009149931371212006, "reward_total_mean": 0.9714630842208862, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9714630842208862, "rewards/meter/std": 0.009149931371212006, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9714630842208862, "rewards/total_composite/std": 0.009149931371212006, "sampling/importance_sampling_ratio/max": 1.4535624980926514, "sampling/importance_sampling_ratio/mean": 1.0066375732421875, "sampling/importance_sampling_ratio/min": 0.5366069078445435, "sampling/sampling_logp_difference/max": 0.6224894523620605, "sampling/sampling_logp_difference/mean": 0.01836930215358734, "step": 1388 }, { "clip_ratio/high_max": 0.009672214509919286, "clip_ratio/high_mean": 0.009672214509919286, "clip_ratio/low_mean": 0.010067827010061592, "clip_ratio/low_min": 0.010067827010061592, "clip_ratio/region_mean": 0.019740041519980878, "completions/clipped_ratio": 0.0, "completions/max_length": 295.0, "completions/max_terminated_length": 295.0, "completions/mean_length": 276.375, "completions/mean_terminated_length": 276.375, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.1869909018278122, "epoch": 0.055789854199301124, "frac_reward_zero_std": 0.0, "grad_norm": 2.405141592025757, "learning_rate": 5.793939393939394e-06, "loss": -0.0159, "num_tokens": 3131978.0, "reward": 0.5435611009597778, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949830770492554, "reward_meter_std": 0.0029864103998988867, "reward_repeat_penalty_mean": 0.7510416507720947, "reward_repeat_penalty_std": 0.10675939172506332, "reward_std": 0.07788533717393875, "reward_total_composite_mean": 0.5435611009597778, "reward_total_composite_std": 0.07788535207509995, "reward_total_mean": 0.5435611009597778, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949830770492554, "rewards/meter/std": 0.0029864103998988867, "rewards/repeat_penalty/mean": 0.7510416507720947, "rewards/repeat_penalty/std": 0.10675939172506332, "rewards/total_composite/mean": 0.5435611009597778, "rewards/total_composite/std": 0.07788535207509995, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002191185951233, "sampling/importance_sampling_ratio/min": 0.1883632391691208, "sampling/sampling_logp_difference/max": 1.6693830490112305, "sampling/sampling_logp_difference/mean": 0.028575100004673004, "step": 1389 }, { "clip_ratio/high_max": 0.011374081019312143, "clip_ratio/high_mean": 0.011374081019312143, "clip_ratio/low_mean": 0.01820334349758923, "clip_ratio/low_min": 0.01820334349758923, "clip_ratio/region_mean": 0.029577424516901374, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 67.5, "completions/mean_terminated_length": 67.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.28015392273664474, "epoch": 0.05583001968108608, "frac_reward_zero_std": 0.0, "grad_norm": 6.391792297363281, "learning_rate": 5.790909090909091e-06, "loss": 0.0238, "num_tokens": 3133718.0, "reward": 0.11840285360813141, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.11840285360813141, "reward_meter_std": 0.15882401168346405, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.15882402658462524, "reward_total_composite_mean": 0.11840285360813141, "reward_total_composite_std": 0.15882401168346405, "reward_total_mean": 0.11840285360813141, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.11840285360813141, "rewards/meter/std": 0.15882401168346405, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.11840285360813141, "rewards/total_composite/std": 0.15882401168346405, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.00211763381958, "sampling/importance_sampling_ratio/min": 0.2688441574573517, "sampling/sampling_logp_difference/max": 1.3136234283447266, "sampling/sampling_logp_difference/mean": 0.04550213739275932, "step": 1390 }, { "clip_ratio/high_max": 0.04821625351905823, "clip_ratio/high_mean": 0.04821625351905823, "clip_ratio/low_mean": 0.005090707214549184, "clip_ratio/low_min": 0.005090707214549184, "clip_ratio/region_mean": 0.05330696073360741, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 76.25, "completions/mean_terminated_length": 76.25, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.3864719457924366, "epoch": 0.05587018516287103, "frac_reward_zero_std": 0.0, "grad_norm": 5.131795406341553, "learning_rate": 5.787878787878788e-06, "loss": -0.0043, "num_tokens": 3135632.0, "reward": 0.9707291722297668, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9707291722297668, "reward_meter_std": 0.025657914578914642, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02565791644155979, "reward_total_composite_mean": 0.9707291722297668, "reward_total_composite_std": 0.025657914578914642, "reward_total_mean": 0.9707291722297668, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9707291722297668, "rewards/meter/std": 0.025657914578914642, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9707291722297668, "rewards/total_composite/std": 0.025657914578914642, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9987998008728027, "sampling/importance_sampling_ratio/min": 0.1429203599691391, "sampling/sampling_logp_difference/max": 1.9454677104949951, "sampling/sampling_logp_difference/mean": 0.0554119348526001, "step": 1391 }, { "clip_ratio/high_max": 0.041855036513879895, "clip_ratio/high_mean": 0.041855036513879895, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.04553150711581111, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 65.125, "completions/mean_terminated_length": 65.125, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2185972724109888, "epoch": 0.055910350644655986, "frac_reward_zero_std": 0.0, "grad_norm": 11.628338813781738, "learning_rate": 5.784848484848486e-06, "loss": 0.0218, "num_tokens": 3137481.0, "reward": 0.9977819919586182, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977819919586182, "reward_meter_std": 0.0022414319682866335, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0022414233535528183, "reward_total_composite_mean": 0.9977819919586182, "reward_total_composite_std": 0.0022414319682866335, "reward_total_mean": 0.9977819919586182, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977819919586182, "rewards/meter/std": 0.0022414319682866335, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977819919586182, "rewards/total_composite/std": 0.0022414319682866335, "sampling/importance_sampling_ratio/max": 1.7962960004806519, "sampling/importance_sampling_ratio/mean": 1.0126001834869385, "sampling/importance_sampling_ratio/min": 0.4376833438873291, "sampling/sampling_logp_difference/max": 0.8262596130371094, "sampling/sampling_logp_difference/mean": 0.035027600824832916, "step": 1392 }, { "clip_ratio/high_max": 0.032872630283236504, "clip_ratio/high_mean": 0.032872630283236504, "clip_ratio/low_mean": 0.03256637742742896, "clip_ratio/low_min": 0.03256637742742896, "clip_ratio/region_mean": 0.06543900771066546, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 58.375, "completions/mean_terminated_length": 58.375, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.6722332537174225, "epoch": 0.05595051612644094, "frac_reward_zero_std": 0.0, "grad_norm": 8.172205924987793, "learning_rate": 5.781818181818181e-06, "loss": -0.0412, "num_tokens": 3139228.0, "reward": 0.45941346883773804, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.48125261068344116, "reward_meter_std": 0.2949608266353607, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.29778623580932617, "reward_total_composite_mean": 0.45941346883773804, "reward_total_composite_std": 0.29778623580932617, "reward_total_mean": 0.45941346883773804, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.48125261068344116, "rewards/meter/std": 0.2949608266353607, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.45941346883773804, "rewards/total_composite/std": 0.29778623580932617, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0176113843917847, "sampling/importance_sampling_ratio/min": 0.2860074043273926, "sampling/sampling_logp_difference/max": 1.2517375946044922, "sampling/sampling_logp_difference/mean": 0.07994799315929413, "step": 1393 }, { "clip_ratio/high_max": 0.01639902195893228, "clip_ratio/high_mean": 0.01639902195893228, "clip_ratio/low_mean": 0.011919185984879732, "clip_ratio/low_min": 0.011919185984879732, "clip_ratio/region_mean": 0.028318207943812013, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 75.625, "completions/mean_terminated_length": 75.625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.136459331959486, "epoch": 0.055990681608225894, "frac_reward_zero_std": 0.0, "grad_norm": 5.719612121582031, "learning_rate": 5.7787878787878795e-06, "loss": 0.1048, "num_tokens": 3141241.0, "reward": 0.4432606101036072, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5849511027336121, "reward_meter_std": 0.4843460023403168, "reward_repeat_penalty_mean": 0.824999988079071, "reward_repeat_penalty_std": 0.19820624589920044, "reward_std": 0.39577043056488037, "reward_total_composite_mean": 0.4432606101036072, "reward_total_composite_std": 0.39577049016952515, "reward_total_mean": 0.4432606101036072, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5849511027336121, "rewards/meter/std": 0.4843460023403168, "rewards/repeat_penalty/mean": 0.824999988079071, "rewards/repeat_penalty/std": 0.19820624589920044, "rewards/total_composite/mean": 0.4432606101036072, "rewards/total_composite/std": 0.39577049016952515, "sampling/importance_sampling_ratio/max": 1.5665262937545776, "sampling/importance_sampling_ratio/mean": 1.0027085542678833, "sampling/importance_sampling_ratio/min": 0.41068050265312195, "sampling/sampling_logp_difference/max": 0.8899397850036621, "sampling/sampling_logp_difference/mean": 0.023130640387535095, "step": 1394 }, { "clip_ratio/high_max": 0.03542683261912316, "clip_ratio/high_mean": 0.03542683261912316, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/region_mean": 0.0372125469148159, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 69.875, "completions/mean_terminated_length": 69.875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.37176400050520897, "epoch": 0.05603084709001085, "frac_reward_zero_std": 0.0, "grad_norm": 4.375842094421387, "learning_rate": 5.775757575757577e-06, "loss": 0.0056, "num_tokens": 3143152.0, "reward": 0.9529762864112854, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9944968223571777, "reward_meter_std": 0.0024136791471391916, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11665360629558563, "reward_total_composite_mean": 0.9529762864112854, "reward_total_composite_std": 0.11665359884500504, "reward_total_mean": 0.9529762864112854, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9944968223571777, "rewards/meter/std": 0.0024136791471391916, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9529762864112854, "rewards/total_composite/std": 0.11665359884500504, "sampling/importance_sampling_ratio/max": 1.78848135471344, "sampling/importance_sampling_ratio/mean": 1.009644627571106, "sampling/importance_sampling_ratio/min": 0.17663374543190002, "sampling/sampling_logp_difference/max": 1.7336769104003906, "sampling/sampling_logp_difference/mean": 0.04802675172686577, "step": 1395 }, { "clip_ratio/high_max": 0.012935725739225745, "clip_ratio/high_mean": 0.012935725739225745, "clip_ratio/low_mean": 0.010312073398381472, "clip_ratio/low_min": 0.010312073398381472, "clip_ratio/region_mean": 0.023247799137607217, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 316.875, "completions/mean_terminated_length": 316.875, "completions/min_length": 296.0, "completions/min_terminated_length": 296.0, "entropy": 0.0825919434428215, "epoch": 0.0560710125717958, "frac_reward_zero_std": 0.0, "grad_norm": 2.749262809753418, "learning_rate": 5.772727272727273e-06, "loss": -0.0037, "num_tokens": 3147511.0, "reward": 0.13397403061389923, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6111111044883728, "reward_count_adherence_std": 0.029695691540837288, "reward_meter_mean": 0.323222815990448, "reward_meter_std": 0.21111111342906952, "reward_repeat_penalty_mean": 0.7082297801971436, "reward_repeat_penalty_std": 0.0814073234796524, "reward_std": 0.07767514139413834, "reward_total_composite_mean": 0.13397403061389923, "reward_total_composite_std": 0.07767514884471893, "reward_total_mean": 0.13397403061389923, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6111111044883728, "rewards/count_adherence/std": 0.029695691540837288, "rewards/meter/mean": 0.323222815990448, "rewards/meter/std": 0.21111111342906952, "rewards/repeat_penalty/mean": 0.7082297801971436, "rewards/repeat_penalty/std": 0.0814073234796524, "rewards/total_composite/mean": 0.13397403061389923, "rewards/total_composite/std": 0.07767514884471893, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9997315406799316, "sampling/importance_sampling_ratio/min": 0.016479218378663063, "sampling/sampling_logp_difference/max": 4.105655193328857, "sampling/sampling_logp_difference/mean": 0.027990715578198433, "step": 1396 }, { "clip_ratio/high_max": 0.00919352809432894, "clip_ratio/high_mean": 0.00919352809432894, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/region_mean": 0.012618185603059828, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.25, "completions/mean_terminated_length": 70.25, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.08694514073431492, "epoch": 0.056111178053580756, "frac_reward_zero_std": 0.0, "grad_norm": 4.050926208496094, "learning_rate": 5.76969696969697e-06, "loss": 0.0166, "num_tokens": 3149369.0, "reward": 0.753129243850708, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9413079023361206, "reward_meter_std": 0.009388444945216179, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.10083267837762833, "reward_total_composite_mean": 0.753129243850708, "reward_total_composite_std": 0.10083267837762833, "reward_total_mean": 0.753129243850708, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9413079023361206, "rewards/meter/std": 0.009388444945216179, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.753129243850708, "rewards/total_composite/std": 0.10083267837762833, "sampling/importance_sampling_ratio/max": 1.494532585144043, "sampling/importance_sampling_ratio/mean": 0.9976539611816406, "sampling/importance_sampling_ratio/min": 0.44028440117836, "sampling/sampling_logp_difference/max": 0.8203344345092773, "sampling/sampling_logp_difference/mean": 0.018803365528583527, "step": 1397 }, { "clip_ratio/high_max": 0.0061509020160883665, "clip_ratio/high_mean": 0.0061509020160883665, "clip_ratio/low_mean": 0.012992122676223516, "clip_ratio/low_min": 0.012992122676223516, "clip_ratio/region_mean": 0.019143024692311883, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 58.875, "completions/mean_terminated_length": 58.875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.17077440861612558, "epoch": 0.05615134353536571, "frac_reward_zero_std": 0.0, "grad_norm": 4.091942310333252, "learning_rate": 5.766666666666667e-06, "loss": -0.0167, "num_tokens": 3151224.0, "reward": 0.6508887410163879, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6508887410163879, "reward_meter_std": 0.1606731116771698, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1606731414794922, "reward_total_composite_mean": 0.6508887410163879, "reward_total_composite_std": 0.1606731116771698, "reward_total_mean": 0.6508887410163879, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6508887410163879, "rewards/meter/std": 0.1606731116771698, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6508887410163879, "rewards/total_composite/std": 0.1606731116771698, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0124748945236206, "sampling/importance_sampling_ratio/min": 0.4910596013069153, "sampling/sampling_logp_difference/max": 0.8628159761428833, "sampling/sampling_logp_difference/mean": 0.02822190895676613, "step": 1398 }, { "clip_ratio/high_max": 0.011937764240428805, "clip_ratio/high_mean": 0.011937764240428805, "clip_ratio/low_mean": 0.00562528264708817, "clip_ratio/low_min": 0.00562528264708817, "clip_ratio/region_mean": 0.017563046887516975, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 63.75, "completions/mean_terminated_length": 63.75, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.1746031977236271, "epoch": 0.05619150901715066, "frac_reward_zero_std": 0.0, "grad_norm": 7.159199237823486, "learning_rate": 5.763636363636365e-06, "loss": 0.0332, "num_tokens": 3152918.0, "reward": 0.9983058571815491, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983058571815491, "reward_meter_std": 0.00029361830092966557, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000293627759674564, "reward_total_composite_mean": 0.9983058571815491, "reward_total_composite_std": 0.00029361830092966557, "reward_total_mean": 0.9983058571815491, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983058571815491, "rewards/meter/std": 0.00029361830092966557, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983058571815491, "rewards/total_composite/std": 0.00029361830092966557, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0061930418014526, "sampling/importance_sampling_ratio/min": 0.13281607627868652, "sampling/sampling_logp_difference/max": 2.0187900066375732, "sampling/sampling_logp_difference/mean": 0.03937718644738197, "step": 1399 }, { "clip_ratio/high_max": 0.027914301492273808, "clip_ratio/high_mean": 0.027914301492273808, "clip_ratio/low_mean": 0.022338663460686803, "clip_ratio/low_min": 0.022338663460686803, "clip_ratio/region_mean": 0.05025296495296061, "completions/clipped_ratio": 0.0, "completions/max_length": 171.0, "completions/max_terminated_length": 171.0, "completions/mean_length": 165.5, "completions/mean_terminated_length": 165.5, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.4123094528913498, "epoch": 0.05623167449893562, "frac_reward_zero_std": 0.0, "grad_norm": 3.5979604721069336, "learning_rate": 5.760606060606061e-06, "loss": 0.0063, "num_tokens": 3155594.0, "reward": 0.7801923751831055, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8759403228759766, "reward_meter_std": 0.19313931465148926, "reward_repeat_penalty_mean": 0.9027777910232544, "reward_repeat_penalty_std": 0.0927247703075409, "reward_std": 0.1476370245218277, "reward_total_composite_mean": 0.7801923751831055, "reward_total_composite_std": 0.1476370245218277, "reward_total_mean": 0.7801923751831055, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8759403228759766, "rewards/meter/std": 0.19313931465148926, "rewards/repeat_penalty/mean": 0.9027777910232544, "rewards/repeat_penalty/std": 0.0927247703075409, "rewards/total_composite/mean": 0.7801923751831055, "rewards/total_composite/std": 0.1476370245218277, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0012664794921875, "sampling/importance_sampling_ratio/min": 0.18697118759155273, "sampling/sampling_logp_difference/max": 1.6768007278442383, "sampling/sampling_logp_difference/mean": 0.05515680089592934, "step": 1400 }, { "epoch": 0.05623167449893562, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 281.6923076923077, "eval_completions/max_terminated_length": 281.6923076923077, "eval_completions/mean_length": 158.23076923076923, "eval_completions/mean_terminated_length": 158.23076923076923, "eval_completions/min_length": 57.92307692307692, "eval_completions/min_terminated_length": 57.92307692307692, "eval_entropy": 0.18456935252134615, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 3155594.0, "eval_reward": 0.42582815427046555, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.8451297649970422, "eval_reward_count_adherence_std": 0.1567458017514302, "eval_reward_meter_mean": 0.6238016211069547, "eval_reward_meter_std": 0.41127324333557713, "eval_reward_repeat_penalty_mean": 0.7824494563616239, "eval_reward_repeat_penalty_std": 0.1815235666357554, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.42582815427046555, "eval_reward_total_composite_std": 0.3518183070879716, "eval_reward_total_mean": 0.42582815427046555, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.8451297649970422, "eval_rewards/count_adherence/std": 0.1567458017514302, "eval_rewards/meter/mean": 0.6238016211069547, "eval_rewards/meter/std": 0.41127324333557713, "eval_rewards/repeat_penalty/mean": 0.7824494563616239, "eval_rewards/repeat_penalty/std": 0.1815235666357554, "eval_rewards/total_composite/mean": 0.42582815427046555, "eval_rewards/total_composite/std": 0.3518183070879716, "eval_runtime": 55.0853, "eval_samples_per_second": 1.888, "eval_sampling/importance_sampling_ratio/max": 1.4843606765453632, "eval_sampling/importance_sampling_ratio/mean": 1.0046258247815645, "eval_sampling/importance_sampling_ratio/min": 0.40005409259062547, "eval_sampling/sampling_logp_difference/max": 0.9362153823559101, "eval_sampling/sampling_logp_difference/mean": 0.02079938738965071, "eval_steps_per_second": 0.236, "step": 1400 }, { "clip_ratio/high_max": 0.014245107769966125, "clip_ratio/high_mean": 0.014245107769966125, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/region_mean": 0.0182773657143116, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.625, "completions/mean_terminated_length": 61.625, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.07412398885935545, "epoch": 0.05627183998072057, "frac_reward_zero_std": 0.0, "grad_norm": 2.925625801086426, "learning_rate": 5.7575757575757586e-06, "loss": 0.0053, "num_tokens": 3157247.0, "reward": 0.9903584122657776, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9903584122657776, "reward_meter_std": 0.011271141469478607, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011271136812865734, "reward_total_composite_mean": 0.9903584122657776, "reward_total_composite_std": 0.011271141469478607, "reward_total_mean": 0.9903584122657776, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9903584122657776, "rewards/meter/std": 0.011271141469478607, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9903584122657776, "rewards/total_composite/std": 0.011271141469478607, "sampling/importance_sampling_ratio/max": 1.5843480825424194, "sampling/importance_sampling_ratio/mean": 1.0016825199127197, "sampling/importance_sampling_ratio/min": 0.22857964038848877, "sampling/sampling_logp_difference/max": 1.4758706092834473, "sampling/sampling_logp_difference/mean": 0.015541310422122478, "step": 1401 }, { "clip_ratio/high_max": 0.0025862068869173527, "clip_ratio/high_mean": 0.0025862068869173527, "clip_ratio/low_mean": 0.013523990812245756, "clip_ratio/low_min": 0.013523990812245756, "clip_ratio/region_mean": 0.01611019769916311, "completions/clipped_ratio": 0.0, "completions/max_length": 147.0, "completions/max_terminated_length": 147.0, "completions/mean_length": 139.0, "completions/mean_terminated_length": 139.0, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.15692214109003544, "epoch": 0.056312005462505525, "frac_reward_zero_std": 0.0, "grad_norm": 2.7992911338806152, "learning_rate": 5.754545454545455e-06, "loss": -0.0084, "num_tokens": 3160271.0, "reward": 0.1747434139251709, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4930988550186157, "reward_meter_std": 0.10791701823472977, "reward_repeat_penalty_mean": 0.4305555522441864, "reward_repeat_penalty_std": 0.15068919956684113, "reward_std": 0.07833553850650787, "reward_total_composite_mean": 0.1747434139251709, "reward_total_composite_std": 0.07833553850650787, "reward_total_mean": 0.1747434139251709, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4930988550186157, "rewards/meter/std": 0.10791701823472977, "rewards/repeat_penalty/mean": 0.4305555522441864, "rewards/repeat_penalty/std": 0.15068919956684113, "rewards/total_composite/mean": 0.1747434139251709, "rewards/total_composite/std": 0.07833553850650787, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0046955347061157, "sampling/importance_sampling_ratio/min": 0.23292852938175201, "sampling/sampling_logp_difference/max": 1.4570236206054688, "sampling/sampling_logp_difference/mean": 0.02746950089931488, "step": 1402 }, { "clip_ratio/high_max": 0.007620060350745916, "clip_ratio/high_mean": 0.007620060350745916, "clip_ratio/low_mean": 0.0072619569837115705, "clip_ratio/low_min": 0.0072619569837115705, "clip_ratio/region_mean": 0.014882017334457487, "completions/clipped_ratio": 0.0, "completions/max_length": 202.0, "completions/max_terminated_length": 202.0, "completions/mean_length": 187.25, "completions/mean_terminated_length": 187.25, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "entropy": 0.19072475843131542, "epoch": 0.05635217094429048, "frac_reward_zero_std": 0.0, "grad_norm": 3.119965076446533, "learning_rate": 5.751515151515152e-06, "loss": -0.0093, "num_tokens": 3163193.0, "reward": 0.7755063772201538, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.958865761756897, "reward_meter_std": 0.024461213499307632, "reward_repeat_penalty_mean": 0.829365074634552, "reward_repeat_penalty_std": 0.10101525485515594, "reward_std": 0.1160004660487175, "reward_total_composite_mean": 0.7755063772201538, "reward_total_composite_std": 0.11600048094987869, "reward_total_mean": 0.7755063772201538, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.958865761756897, "rewards/meter/std": 0.024461213499307632, "rewards/repeat_penalty/mean": 0.829365074634552, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.7755063772201538, "rewards/total_composite/std": 0.11600048094987869, "sampling/importance_sampling_ratio/max": 1.7141207456588745, "sampling/importance_sampling_ratio/mean": 1.0046159029006958, "sampling/importance_sampling_ratio/min": 0.2828236222267151, "sampling/sampling_logp_difference/max": 1.2629318237304688, "sampling/sampling_logp_difference/mean": 0.030590685084462166, "step": 1403 }, { "clip_ratio/high_max": 0.019914651405997574, "clip_ratio/high_mean": 0.019914651405997574, "clip_ratio/low_mean": 0.0077174786711111665, "clip_ratio/low_min": 0.0077174786711111665, "clip_ratio/region_mean": 0.02763213007710874, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 77.75, "completions/mean_terminated_length": 77.75, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.3366150539368391, "epoch": 0.05639233642607543, "frac_reward_zero_std": 0.0, "grad_norm": 4.241501808166504, "learning_rate": 5.748484848484849e-06, "loss": 0.0225, "num_tokens": 3165135.0, "reward": 0.9753485321998596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9753485321998596, "reward_meter_std": 0.02492102049291134, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.024921026080846786, "reward_total_composite_mean": 0.9753485321998596, "reward_total_composite_std": 0.02492102049291134, "reward_total_mean": 0.9753485321998596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9753485321998596, "rewards/meter/std": 0.02492102049291134, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9753485321998596, "rewards/total_composite/std": 0.02492102049291134, "sampling/importance_sampling_ratio/max": 1.8341128826141357, "sampling/importance_sampling_ratio/mean": 1.0083844661712646, "sampling/importance_sampling_ratio/min": 0.21236151456832886, "sampling/sampling_logp_difference/max": 1.5494651794433594, "sampling/sampling_logp_difference/mean": 0.04453163966536522, "step": 1404 }, { "clip_ratio/high_max": 0.0202326342696324, "clip_ratio/high_mean": 0.0202326342696324, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.022070869570598006, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.375, "completions/mean_terminated_length": 67.375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.20304011879488826, "epoch": 0.05643250190786039, "frac_reward_zero_std": 0.0, "grad_norm": 3.760753631591797, "learning_rate": 5.745454545454546e-06, "loss": 0.0037, "num_tokens": 3166898.0, "reward": 0.9491584897041321, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9491584897041321, "reward_meter_std": 0.09555412083864212, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09555413573980331, "reward_total_composite_mean": 0.9491584897041321, "reward_total_composite_std": 0.09555412083864212, "reward_total_mean": 0.9491584897041321, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9491584897041321, "rewards/meter/std": 0.09555412083864212, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9491584897041321, "rewards/total_composite/std": 0.09555412083864212, "sampling/importance_sampling_ratio/max": 1.8443472385406494, "sampling/importance_sampling_ratio/mean": 1.005703091621399, "sampling/importance_sampling_ratio/min": 0.26183465123176575, "sampling/sampling_logp_difference/max": 1.3400421142578125, "sampling/sampling_logp_difference/mean": 0.026108134537935257, "step": 1405 }, { "clip_ratio/high_max": 0.03157911077141762, "clip_ratio/high_mean": 0.03157911077141762, "clip_ratio/low_mean": 0.025304300244897604, "clip_ratio/low_min": 0.025304300244897604, "clip_ratio/region_mean": 0.05688341101631522, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 59.25, "completions/mean_terminated_length": 59.25, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.4192809797823429, "epoch": 0.05647266738964534, "frac_reward_zero_std": 0.0, "grad_norm": 5.033778190612793, "learning_rate": 5.742424242424242e-06, "loss": 0.0102, "num_tokens": 3168628.0, "reward": 0.37988948822021484, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.37988948822021484, "reward_meter_std": 0.33830785751342773, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.33830785751342773, "reward_total_composite_mean": 0.37988948822021484, "reward_total_composite_std": 0.33830785751342773, "reward_total_mean": 0.37988948822021484, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.37988948822021484, "rewards/meter/std": 0.33830785751342773, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.37988948822021484, "rewards/total_composite/std": 0.33830785751342773, "sampling/importance_sampling_ratio/max": 1.7001794576644897, "sampling/importance_sampling_ratio/mean": 1.01566481590271, "sampling/importance_sampling_ratio/min": 0.22231467068195343, "sampling/sampling_logp_difference/max": 1.5036615133285522, "sampling/sampling_logp_difference/mean": 0.05973369628190994, "step": 1406 }, { "clip_ratio/high_max": 0.0139403420034796, "clip_ratio/high_mean": 0.0139403420034796, "clip_ratio/low_mean": 0.01607399620115757, "clip_ratio/low_min": 0.01607399620115757, "clip_ratio/region_mean": 0.03001433820463717, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 99.375, "completions/mean_terminated_length": 99.375, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.32133867032825947, "epoch": 0.056512832871430295, "frac_reward_zero_std": 0.0, "grad_norm": 6.2097625732421875, "learning_rate": 5.73939393939394e-06, "loss": 0.0163, "num_tokens": 3170743.0, "reward": 0.829431414604187, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8928050398826599, "reward_meter_std": 0.15822991728782654, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.1831192672252655, "reward_total_composite_mean": 0.829431414604187, "reward_total_composite_std": 0.1831192672252655, "reward_total_mean": 0.829431414604187, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8928050398826599, "rewards/meter/std": 0.15822991728782654, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.829431414604187, "rewards/total_composite/std": 0.1831192672252655, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0000447034835815, "sampling/importance_sampling_ratio/min": 0.2882235050201416, "sampling/sampling_logp_difference/max": 1.2440190315246582, "sampling/sampling_logp_difference/mean": 0.04797036573290825, "step": 1407 }, { "clip_ratio/high_max": 0.01601142482832074, "clip_ratio/high_mean": 0.01601142482832074, "clip_ratio/low_mean": 0.008198924828320742, "clip_ratio/low_min": 0.008198924828320742, "clip_ratio/region_mean": 0.024210349656641483, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 30.875, "completions/mean_terminated_length": 30.875, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.1420626100152731, "epoch": 0.05655299835321525, "frac_reward_zero_std": 0.0, "grad_norm": 7.772467136383057, "learning_rate": 5.736363636363637e-06, "loss": -0.0044, "num_tokens": 3172326.0, "reward": 0.9951827526092529, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9951827526092529, "reward_meter_std": 0.007008199580013752, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007008187007158995, "reward_total_composite_mean": 0.9951827526092529, "reward_total_composite_std": 0.007008199580013752, "reward_total_mean": 0.9951827526092529, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9951827526092529, "rewards/meter/std": 0.007008199580013752, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951827526092529, "rewards/total_composite/std": 0.007008199580013752, "sampling/importance_sampling_ratio/max": 1.774673581123352, "sampling/importance_sampling_ratio/mean": 0.9993618726730347, "sampling/importance_sampling_ratio/min": 0.4305649697780609, "sampling/sampling_logp_difference/max": 0.8426570892333984, "sampling/sampling_logp_difference/mean": 0.033370744436979294, "step": 1408 }, { "clip_ratio/high_max": 0.016995614394545555, "clip_ratio/high_mean": 0.016995614394545555, "clip_ratio/low_mean": 0.02108370151836425, "clip_ratio/low_min": 0.02108370151836425, "clip_ratio/region_mean": 0.038079315912909806, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 51.625, "completions/mean_terminated_length": 51.625, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.28332205303013325, "epoch": 0.0565931638350002, "frac_reward_zero_std": 0.0, "grad_norm": 11.023374557495117, "learning_rate": 5.733333333333334e-06, "loss": 0.2629, "num_tokens": 3173907.0, "reward": 0.4936572313308716, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.5345224738121033, "reward_meter_mean": 0.992419958114624, "reward_meter_std": 0.00853477232158184, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.5277819633483887, "reward_total_composite_mean": 0.4936572313308716, "reward_total_composite_std": 0.5277820229530334, "reward_total_mean": 0.4936572313308716, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.5345224738121033, "rewards/meter/mean": 0.992419958114624, "rewards/meter/std": 0.00853477232158184, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4936572313308716, "rewards/total_composite/std": 0.5277820229530334, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9999293088912964, "sampling/importance_sampling_ratio/min": 0.018808001652359962, "sampling/sampling_logp_difference/max": 3.973472833633423, "sampling/sampling_logp_difference/mean": 0.06906002014875412, "step": 1409 }, { "clip_ratio/high_max": 0.04356799623928964, "clip_ratio/high_mean": 0.04356799623928964, "clip_ratio/low_mean": 0.01024590153247118, "clip_ratio/low_min": 0.01024590153247118, "clip_ratio/region_mean": 0.05381389777176082, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.20567422546446323, "epoch": 0.05663332931678516, "frac_reward_zero_std": 0.0, "grad_norm": 5.875049591064453, "learning_rate": 5.7303030303030305e-06, "loss": 0.0359, "num_tokens": 3175555.0, "reward": 0.7323773503303528, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7323773503303528, "reward_meter_std": 0.34669190645217896, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.34669187664985657, "reward_total_composite_mean": 0.7323773503303528, "reward_total_composite_std": 0.34669190645217896, "reward_total_mean": 0.7323773503303528, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7323773503303528, "rewards/meter/std": 0.34669190645217896, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7323773503303528, "rewards/total_composite/std": 0.34669190645217896, "sampling/importance_sampling_ratio/max": 1.7754662036895752, "sampling/importance_sampling_ratio/mean": 1.0006393194198608, "sampling/importance_sampling_ratio/min": 0.12341582030057907, "sampling/sampling_logp_difference/max": 2.092195987701416, "sampling/sampling_logp_difference/mean": 0.04662606492638588, "step": 1410 }, { "clip_ratio/high_max": 0.02417309512384236, "clip_ratio/high_mean": 0.02417309512384236, "clip_ratio/low_mean": 0.007942708441987634, "clip_ratio/low_min": 0.007942708441987634, "clip_ratio/region_mean": 0.03211580356582999, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 58.625, "completions/mean_terminated_length": 58.625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.22591773979365826, "epoch": 0.05667349479857011, "frac_reward_zero_std": 0.0, "grad_norm": 7.019533157348633, "learning_rate": 5.727272727272728e-06, "loss": 0.0404, "num_tokens": 3177472.0, "reward": 0.930023729801178, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.930023729801178, "reward_meter_std": 0.11754162609577179, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1175416111946106, "reward_total_composite_mean": 0.930023729801178, "reward_total_composite_std": 0.11754162609577179, "reward_total_mean": 0.930023729801178, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.930023729801178, "rewards/meter/std": 0.11754162609577179, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.930023729801178, "rewards/total_composite/std": 0.11754162609577179, "sampling/importance_sampling_ratio/max": 1.9308364391326904, "sampling/importance_sampling_ratio/mean": 1.0033373832702637, "sampling/importance_sampling_ratio/min": 0.3903082311153412, "sampling/sampling_logp_difference/max": 0.9408185482025146, "sampling/sampling_logp_difference/mean": 0.03719957172870636, "step": 1411 }, { "clip_ratio/high_max": 0.03802949539385736, "clip_ratio/high_mean": 0.03802949539385736, "clip_ratio/low_mean": 0.007936508394777775, "clip_ratio/low_min": 0.007936508394777775, "clip_ratio/region_mean": 0.045966003788635135, "completions/clipped_ratio": 0.0, "completions/max_length": 126.0, "completions/max_terminated_length": 126.0, "completions/mean_length": 121.75, "completions/mean_terminated_length": 121.75, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.25815106742084026, "epoch": 0.056713660280355065, "frac_reward_zero_std": 0.0, "grad_norm": 3.1584293842315674, "learning_rate": 5.724242424242424e-06, "loss": 0.0095, "num_tokens": 3179678.0, "reward": 0.8404378890991211, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.980501115322113, "reward_meter_std": 0.04536837711930275, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.0856877937912941, "reward_total_composite_mean": 0.8404378890991211, "reward_total_composite_std": 0.0856878012418747, "reward_total_mean": 0.8404378890991211, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.980501115322113, "rewards/meter/std": 0.04536837711930275, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8404378890991211, "rewards/total_composite/std": 0.0856878012418747, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0000073909759521, "sampling/importance_sampling_ratio/min": 0.19929902255535126, "sampling/sampling_logp_difference/max": 1.8470945358276367, "sampling/sampling_logp_difference/mean": 0.04456208273768425, "step": 1412 }, { "clip_ratio/high_max": 0.019891314674168825, "clip_ratio/high_mean": 0.019891314674168825, "clip_ratio/low_mean": 0.02131642634049058, "clip_ratio/low_min": 0.02131642634049058, "clip_ratio/region_mean": 0.041207741014659405, "completions/clipped_ratio": 0.0, "completions/max_length": 155.0, "completions/max_terminated_length": 155.0, "completions/mean_length": 148.25, "completions/mean_terminated_length": 148.25, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.24800713174045086, "epoch": 0.05675382576214002, "frac_reward_zero_std": 0.0, "grad_norm": 2.6527252197265625, "learning_rate": 5.721212121212122e-06, "loss": -0.0061, "num_tokens": 3182168.0, "reward": 0.8531917333602905, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9954190850257874, "reward_meter_std": 0.002940161619335413, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.07582316547632217, "reward_total_composite_mean": 0.8531917333602905, "reward_total_composite_std": 0.07582316547632217, "reward_total_mean": 0.8531917333602905, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9954190850257874, "rewards/meter/std": 0.002940161619335413, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8531917333602905, "rewards/total_composite/std": 0.07582316547632217, "sampling/importance_sampling_ratio/max": 1.6937066316604614, "sampling/importance_sampling_ratio/mean": 1.0049986839294434, "sampling/importance_sampling_ratio/min": 0.22713692486286163, "sampling/sampling_logp_difference/max": 1.4822022914886475, "sampling/sampling_logp_difference/mean": 0.04073692858219147, "step": 1413 }, { "clip_ratio/high_max": 0.0170001951046288, "clip_ratio/high_mean": 0.0170001951046288, "clip_ratio/low_mean": 0.007663170341402292, "clip_ratio/low_min": 0.007663170341402292, "clip_ratio/region_mean": 0.024663365446031094, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.16014890000224113, "epoch": 0.05679399124392497, "frac_reward_zero_std": 0.0, "grad_norm": 8.496710777282715, "learning_rate": 5.718181818181819e-06, "loss": -0.0031, "num_tokens": 3183867.0, "reward": 0.7868467569351196, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7868467569351196, "reward_meter_std": 0.3379902243614197, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3379901945590973, "reward_total_composite_mean": 0.7868467569351196, "reward_total_composite_std": 0.3379902243614197, "reward_total_mean": 0.7868467569351196, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7868467569351196, "rewards/meter/std": 0.3379902243614197, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7868467569351196, "rewards/total_composite/std": 0.3379902243614197, "sampling/importance_sampling_ratio/max": 1.4725927114486694, "sampling/importance_sampling_ratio/mean": 0.9952957034111023, "sampling/importance_sampling_ratio/min": 0.13834954798221588, "sampling/sampling_logp_difference/max": 1.9779717922210693, "sampling/sampling_logp_difference/mean": 0.03707931190729141, "step": 1414 }, { "clip_ratio/high_max": 0.04510642075911164, "clip_ratio/high_mean": 0.04510642075911164, "clip_ratio/low_mean": 0.00937500037252903, "clip_ratio/low_min": 0.00937500037252903, "clip_ratio/region_mean": 0.05448142113164067, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 41.0, "completions/mean_terminated_length": 41.0, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.4230155944824219, "epoch": 0.056834156725709926, "frac_reward_zero_std": 0.0, "grad_norm": 12.028426170349121, "learning_rate": 5.715151515151516e-06, "loss": 0.0077, "num_tokens": 3185507.0, "reward": 0.9893122911453247, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9893122911453247, "reward_meter_std": 0.020975705236196518, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.020975695922970772, "reward_total_composite_mean": 0.9893122911453247, "reward_total_composite_std": 0.020975705236196518, "reward_total_mean": 0.9893122911453247, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9893122911453247, "rewards/meter/std": 0.020975705236196518, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9893122911453247, "rewards/total_composite/std": 0.020975705236196518, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0033856630325317, "sampling/importance_sampling_ratio/min": 0.12629108130931854, "sampling/sampling_logp_difference/max": 2.0691659450531006, "sampling/sampling_logp_difference/mean": 0.07620865851640701, "step": 1415 }, { "clip_ratio/high_max": 0.006235292763449252, "clip_ratio/high_mean": 0.006235292763449252, "clip_ratio/low_mean": 0.009044285747222602, "clip_ratio/low_min": 0.009044285747222602, "clip_ratio/region_mean": 0.015279578510671854, "completions/clipped_ratio": 0.0, "completions/max_length": 183.0, "completions/max_terminated_length": 183.0, "completions/mean_length": 179.625, "completions/mean_terminated_length": 179.625, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.15431117359548807, "epoch": 0.05687432220749488, "frac_reward_zero_std": 0.0, "grad_norm": 1.9633679389953613, "learning_rate": 5.712121212121212e-06, "loss": 0.0021, "num_tokens": 3188624.0, "reward": 0.8011291027069092, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946449995040894, "reward_meter_std": 0.006311771925538778, "reward_repeat_penalty_mean": 0.8055555820465088, "reward_repeat_penalty_std": 0.07856741547584534, "reward_std": 0.07706642150878906, "reward_total_composite_mean": 0.8011291027069092, "reward_total_composite_std": 0.07706641405820847, "reward_total_mean": 0.8011291027069092, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946449995040894, "rewards/meter/std": 0.006311771925538778, "rewards/repeat_penalty/mean": 0.8055555820465088, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8011291027069092, "rewards/total_composite/std": 0.07706641405820847, "sampling/importance_sampling_ratio/max": 1.6300792694091797, "sampling/importance_sampling_ratio/mean": 1.0057470798492432, "sampling/importance_sampling_ratio/min": 0.20250163972377777, "sampling/sampling_logp_difference/max": 1.5970072746276855, "sampling/sampling_logp_difference/mean": 0.025626802816987038, "step": 1416 }, { "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/region_mean": 0.006181693868711591, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.125, "completions/mean_terminated_length": 61.125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.06280876183882356, "epoch": 0.056914487689279834, "frac_reward_zero_std": 0.0, "grad_norm": 3.094541549682617, "learning_rate": 5.7090909090909096e-06, "loss": -0.0002, "num_tokens": 3190361.0, "reward": 0.998288631439209, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998288631439209, "reward_meter_std": 0.00019117545161861926, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00019117545161861926, "reward_total_composite_mean": 0.998288631439209, "reward_total_composite_std": 0.00019117545161861926, "reward_total_mean": 0.998288631439209, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998288631439209, "rewards/meter/std": 0.00019117545161861926, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998288631439209, "rewards/total_composite/std": 0.00019117545161861926, "sampling/importance_sampling_ratio/max": 1.6339690685272217, "sampling/importance_sampling_ratio/mean": 1.0027981996536255, "sampling/importance_sampling_ratio/min": 0.26635390520095825, "sampling/sampling_logp_difference/max": 1.3229293823242188, "sampling/sampling_logp_difference/mean": 0.01802898570895195, "step": 1417 }, { "clip_ratio/high_max": 0.013707729522138834, "clip_ratio/high_mean": 0.013707729522138834, "clip_ratio/low_mean": 0.013166894670575857, "clip_ratio/low_min": 0.013166894670575857, "clip_ratio/region_mean": 0.02687462419271469, "completions/clipped_ratio": 0.0, "completions/max_length": 51.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 45.625, "completions/mean_terminated_length": 45.625, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.32396047934889793, "epoch": 0.05695465317106479, "frac_reward_zero_std": 0.0, "grad_norm": 10.003430366516113, "learning_rate": 5.706060606060606e-06, "loss": 0.0502, "num_tokens": 3191990.0, "reward": 0.8391948938369751, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8391948938369751, "reward_meter_std": 0.1604534387588501, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1604534238576889, "reward_total_composite_mean": 0.8391948938369751, "reward_total_composite_std": 0.1604534387588501, "reward_total_mean": 0.8391948938369751, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8391948938369751, "rewards/meter/std": 0.1604534387588501, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8391948938369751, "rewards/total_composite/std": 0.1604534387588501, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0050475597381592, "sampling/importance_sampling_ratio/min": 0.2660462260246277, "sampling/sampling_logp_difference/max": 1.3240852355957031, "sampling/sampling_logp_difference/mean": 0.05166694149374962, "step": 1418 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 31.0, "completions/max_terminated_length": 31.0, "completions/mean_length": 31.0, "completions/mean_terminated_length": 31.0, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.038966658525168896, "epoch": 0.05699481865284974, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.703030303030303e-06, "loss": 0.0, "num_tokens": 3193518.0, "reward": 0.9990130662918091, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990130662918091, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9990130662918091, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9990130662918091, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990130662918091, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990130662918091, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.2404334545135498, "sampling/importance_sampling_ratio/mean": 1.0024038553237915, "sampling/importance_sampling_ratio/min": 0.733427107334137, "sampling/sampling_logp_difference/max": 0.3100270926952362, "sampling/sampling_logp_difference/mean": 0.006462951190769672, "step": 1419 }, { "clip_ratio/high_max": 0.01176445058081299, "clip_ratio/high_mean": 0.01176445058081299, "clip_ratio/low_mean": 0.0031779661076143384, "clip_ratio/low_min": 0.0031779661076143384, "clip_ratio/region_mean": 0.014942416688427329, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 118.375, "completions/mean_terminated_length": 118.375, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.07044978253543377, "epoch": 0.057034984134634696, "frac_reward_zero_std": 0.0, "grad_norm": 3.3092191219329834, "learning_rate": 5.7e-06, "loss": 0.0034, "num_tokens": 3195977.0, "reward": 0.6418830156326294, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984818696975708, "reward_meter_std": 8.360291394637898e-05, "reward_repeat_penalty_mean": 0.8035714030265808, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.05907979980111122, "reward_total_composite_mean": 0.6418830156326294, "reward_total_composite_std": 0.05907980352640152, "reward_total_mean": 0.6418830156326294, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984818696975708, "rewards/meter/std": 8.360291394637898e-05, "rewards/repeat_penalty/mean": 0.8035714030265808, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.6418830156326294, "rewards/total_composite/std": 0.05907980352640152, "sampling/importance_sampling_ratio/max": 1.6658356189727783, "sampling/importance_sampling_ratio/mean": 1.001312494277954, "sampling/importance_sampling_ratio/min": 0.17157158255577087, "sampling/sampling_logp_difference/max": 1.7627546787261963, "sampling/sampling_logp_difference/mean": 0.015416157431900501, "step": 1420 }, { "clip_ratio/high_max": 0.049062950536608696, "clip_ratio/high_mean": 0.049062950536608696, "clip_ratio/low_mean": 0.018345543881878257, "clip_ratio/low_min": 0.018345543881878257, "clip_ratio/region_mean": 0.06740849441848695, "completions/clipped_ratio": 0.0, "completions/max_length": 45.0, "completions/max_terminated_length": 45.0, "completions/mean_length": 42.75, "completions/mean_terminated_length": 42.75, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.3276716824620962, "epoch": 0.05707514961641965, "frac_reward_zero_std": 0.0, "grad_norm": 11.100446701049805, "learning_rate": 5.696969696969698e-06, "loss": -0.006, "num_tokens": 3197831.0, "reward": 0.7552404403686523, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7552404403686523, "reward_meter_std": 0.31532391905784607, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.31532391905784607, "reward_total_composite_mean": 0.7552404403686523, "reward_total_composite_std": 0.31532391905784607, "reward_total_mean": 0.7552404403686523, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7552404403686523, "rewards/meter/std": 0.31532391905784607, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7552404403686523, "rewards/total_composite/std": 0.31532391905784607, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007340669631958, "sampling/importance_sampling_ratio/min": 0.36432281136512756, "sampling/sampling_logp_difference/max": 1.009714961051941, "sampling/sampling_logp_difference/mean": 0.06012430414557457, "step": 1421 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/region_mean": 0.004310344811528921, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 29.375, "completions/mean_terminated_length": 29.375, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.09071109816431999, "epoch": 0.057115315098204604, "frac_reward_zero_std": 0.0, "grad_norm": 3.4191842079162598, "learning_rate": 5.693939393939394e-06, "loss": 0.0018, "num_tokens": 3199274.0, "reward": 0.9963138103485107, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963138103485107, "reward_meter_std": 0.00021444838785100728, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00021445513993967324, "reward_total_composite_mean": 0.9963138103485107, "reward_total_composite_std": 0.00021444838785100728, "reward_total_mean": 0.9963138103485107, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963138103485107, "rewards/meter/std": 0.00021444838785100728, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963138103485107, "rewards/total_composite/std": 0.00021444838785100728, "sampling/importance_sampling_ratio/max": 1.1206741333007812, "sampling/importance_sampling_ratio/mean": 1.001716136932373, "sampling/importance_sampling_ratio/min": 0.5292962193489075, "sampling/sampling_logp_difference/max": 0.6362069845199585, "sampling/sampling_logp_difference/mean": 0.013728239573538303, "step": 1422 }, { "clip_ratio/high_max": 0.014817290706560016, "clip_ratio/high_mean": 0.014817290706560016, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.014817290706560016, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 34.125, "completions/mean_terminated_length": 34.125, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.10061774123460054, "epoch": 0.05715548057998956, "frac_reward_zero_std": 0.0, "grad_norm": 12.835563659667969, "learning_rate": 5.690909090909091e-06, "loss": 0.0189, "num_tokens": 3200795.0, "reward": 0.9920762777328491, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9920762777328491, "reward_meter_std": 0.0020317891612648964, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0020317891612648964, "reward_total_composite_mean": 0.9920762777328491, "reward_total_composite_std": 0.0020317891612648964, "reward_total_mean": 0.9920762777328491, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9920762777328491, "rewards/meter/std": 0.0020317891612648964, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9920762777328491, "rewards/total_composite/std": 0.0020317891612648964, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9959531426429749, "sampling/importance_sampling_ratio/min": 0.47128504514694214, "sampling/sampling_logp_difference/max": 0.7522921562194824, "sampling/sampling_logp_difference/mean": 0.02493356168270111, "step": 1423 }, { "clip_ratio/high_max": 0.0493996306322515, "clip_ratio/high_mean": 0.0493996306322515, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0493996306322515, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 55.5, "completions/mean_terminated_length": 55.5, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.2528437674045563, "epoch": 0.05719564606177451, "frac_reward_zero_std": 0.0, "grad_norm": 5.431046485900879, "learning_rate": 5.687878787878789e-06, "loss": -0.0099, "num_tokens": 3202519.0, "reward": 0.9895903468132019, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9895903468132019, "reward_meter_std": 0.012036402709782124, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.012036396190524101, "reward_total_composite_mean": 0.9895903468132019, "reward_total_composite_std": 0.012036402709782124, "reward_total_mean": 0.9895903468132019, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9895903468132019, "rewards/meter/std": 0.012036402709782124, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9895903468132019, "rewards/total_composite/std": 0.012036402709782124, "sampling/importance_sampling_ratio/max": 1.5429896116256714, "sampling/importance_sampling_ratio/mean": 1.0014959573745728, "sampling/importance_sampling_ratio/min": 0.2612825632095337, "sampling/sampling_logp_difference/max": 1.3421528339385986, "sampling/sampling_logp_difference/mean": 0.044165343046188354, "step": 1424 }, { "clip_ratio/high_max": 0.018805116647854447, "clip_ratio/high_mean": 0.018805116647854447, "clip_ratio/low_mean": 0.006616426864638925, "clip_ratio/low_min": 0.006616426864638925, "clip_ratio/region_mean": 0.025421543512493372, "completions/clipped_ratio": 0.0, "completions/max_length": 331.0, "completions/max_terminated_length": 331.0, "completions/mean_length": 320.375, "completions/mean_terminated_length": 320.375, "completions/min_length": 305.0, "completions/min_terminated_length": 305.0, "entropy": 0.2617227640002966, "epoch": 0.057235811543559466, "frac_reward_zero_std": 0.0, "grad_norm": 2.6007766723632812, "learning_rate": 5.684848484848485e-06, "loss": 0.0056, "num_tokens": 3206882.0, "reward": 0.4912066161632538, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.033064987510442734, "reward_meter_mean": 0.942986011505127, "reward_meter_std": 0.145384281873703, "reward_repeat_penalty_mean": 0.8428308963775635, "reward_repeat_penalty_std": 0.10538350045681, "reward_std": 0.07578601688146591, "reward_total_composite_mean": 0.4912066161632538, "reward_total_composite_std": 0.07578601688146591, "reward_total_mean": 0.4912066161632538, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.033064987510442734, "rewards/meter/mean": 0.942986011505127, "rewards/meter/std": 0.145384281873703, "rewards/repeat_penalty/mean": 0.8428308963775635, "rewards/repeat_penalty/std": 0.10538350045681, "rewards/total_composite/mean": 0.4912066161632538, "rewards/total_composite/std": 0.07578601688146591, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0055261850357056, "sampling/importance_sampling_ratio/min": 0.11778980493545532, "sampling/sampling_logp_difference/max": 2.1388535499572754, "sampling/sampling_logp_difference/mean": 0.03792227804660797, "step": 1425 }, { "clip_ratio/high_max": 0.043533024145290256, "clip_ratio/high_mean": 0.043533024145290256, "clip_ratio/low_mean": 0.006051587639376521, "clip_ratio/low_min": 0.006051587639376521, "clip_ratio/region_mean": 0.04958461178466678, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.2404075190424919, "epoch": 0.05727597702534442, "frac_reward_zero_std": 0.0, "grad_norm": 12.311224937438965, "learning_rate": 5.681818181818183e-06, "loss": -0.01, "num_tokens": 3208730.0, "reward": 0.996849000453949, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996849000453949, "reward_meter_std": 0.0023078599479049444, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002307864371687174, "reward_total_composite_mean": 0.996849000453949, "reward_total_composite_std": 0.0023078599479049444, "reward_total_mean": 0.996849000453949, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996849000453949, "rewards/meter/std": 0.0023078599479049444, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996849000453949, "rewards/total_composite/std": 0.0023078599479049444, "sampling/importance_sampling_ratio/max": 1.7190132141113281, "sampling/importance_sampling_ratio/mean": 1.0055710077285767, "sampling/importance_sampling_ratio/min": 0.15944872796535492, "sampling/sampling_logp_difference/max": 1.8360328674316406, "sampling/sampling_logp_difference/mean": 0.03911422938108444, "step": 1426 }, { "clip_ratio/high_max": 0.012471178779378533, "clip_ratio/high_mean": 0.012471178779378533, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.012471178779378533, "completions/clipped_ratio": 0.0, "completions/max_length": 92.0, "completions/max_terminated_length": 92.0, "completions/mean_length": 90.375, "completions/mean_terminated_length": 90.375, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.06489505246281624, "epoch": 0.057316142507129374, "frac_reward_zero_std": 0.0, "grad_norm": 2.7146718502044678, "learning_rate": 5.67878787878788e-06, "loss": -0.0018, "num_tokens": 3210997.0, "reward": 0.9485726356506348, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984976053237915, "reward_meter_std": 9.82403289526701e-05, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09244248270988464, "reward_total_composite_mean": 0.9485726356506348, "reward_total_composite_std": 0.09244248270988464, "reward_total_mean": 0.9485726356506348, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984976053237915, "rewards/meter/std": 9.82403289526701e-05, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9485726356506348, "rewards/total_composite/std": 0.09244248270988464, "sampling/importance_sampling_ratio/max": 1.3982148170471191, "sampling/importance_sampling_ratio/mean": 0.9975816011428833, "sampling/importance_sampling_ratio/min": 0.33406925201416016, "sampling/sampling_logp_difference/max": 1.0964069366455078, "sampling/sampling_logp_difference/mean": 0.014895378611981869, "step": 1427 }, { "clip_ratio/high_max": 0.012853078544139862, "clip_ratio/high_mean": 0.012853078544139862, "clip_ratio/low_mean": 0.01068126189056784, "clip_ratio/low_min": 0.01068126189056784, "clip_ratio/region_mean": 0.0235343404347077, "completions/clipped_ratio": 0.0, "completions/max_length": 272.0, "completions/max_terminated_length": 272.0, "completions/mean_length": 256.0, "completions/mean_terminated_length": 256.0, "completions/min_length": 245.0, "completions/min_terminated_length": 245.0, "entropy": 0.24354491382837296, "epoch": 0.05735630798891433, "frac_reward_zero_std": 0.0, "grad_norm": 2.3229403495788574, "learning_rate": 5.675757575757577e-06, "loss": 0.0093, "num_tokens": 3214669.0, "reward": 0.5593699216842651, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.699999988079071, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9840718507766724, "reward_meter_std": 0.03672640025615692, "reward_repeat_penalty_mean": 0.8125, "reward_repeat_penalty_std": 0.10492872446775436, "reward_std": 0.07350601255893707, "reward_total_composite_mean": 0.5593699216842651, "reward_total_composite_std": 0.07350600510835648, "reward_total_mean": 0.5593699216842651, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.699999988079071, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9840718507766724, "rewards/meter/std": 0.03672640025615692, "rewards/repeat_penalty/mean": 0.8125, "rewards/repeat_penalty/std": 0.10492872446775436, "rewards/total_composite/mean": 0.5593699216842651, "rewards/total_composite/std": 0.07350600510835648, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0045963525772095, "sampling/importance_sampling_ratio/min": 0.035527586936950684, "sampling/sampling_logp_difference/max": 3.3374457359313965, "sampling/sampling_logp_difference/mean": 0.0390043631196022, "step": 1428 }, { "clip_ratio/high_max": 0.013211918994784355, "clip_ratio/high_mean": 0.013211918994784355, "clip_ratio/low_mean": 0.006106290849857032, "clip_ratio/low_min": 0.006106290849857032, "clip_ratio/region_mean": 0.019318209844641387, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 123.25, "completions/mean_terminated_length": 123.25, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.07903782464563847, "epoch": 0.05739647347069928, "frac_reward_zero_std": 0.0, "grad_norm": 2.9396092891693115, "learning_rate": 5.672727272727273e-06, "loss": 0.0002, "num_tokens": 3217095.0, "reward": 0.8018503189086914, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978071451187134, "reward_meter_std": 0.0011365336831659079, "reward_repeat_penalty_mean": 0.8035714030265808, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07426347583532333, "reward_total_composite_mean": 0.8018503189086914, "reward_total_composite_std": 0.07426349073648453, "reward_total_mean": 0.8018503189086914, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978071451187134, "rewards/meter/std": 0.0011365336831659079, "rewards/repeat_penalty/mean": 0.8035714030265808, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.8018503189086914, "rewards/total_composite/std": 0.07426349073648453, "sampling/importance_sampling_ratio/max": 1.302078366279602, "sampling/importance_sampling_ratio/mean": 0.999223530292511, "sampling/importance_sampling_ratio/min": 0.2165946513414383, "sampling/sampling_logp_difference/max": 1.5297276973724365, "sampling/sampling_logp_difference/mean": 0.015226030722260475, "step": 1429 }, { "clip_ratio/high_max": 0.008850250858813524, "clip_ratio/high_mean": 0.008850250858813524, "clip_ratio/low_mean": 0.030917756259441376, "clip_ratio/low_min": 0.030917756259441376, "clip_ratio/region_mean": 0.0397680071182549, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 56.5, "completions/mean_terminated_length": 56.5, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.5998508296906948, "epoch": 0.057436638952484236, "frac_reward_zero_std": 0.0, "grad_norm": 5.2827301025390625, "learning_rate": 5.6696969696969705e-06, "loss": 0.0069, "num_tokens": 3218659.0, "reward": 0.43106603622436523, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.43106603622436523, "reward_meter_std": 0.3671012222766876, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.36710119247436523, "reward_total_composite_mean": 0.43106603622436523, "reward_total_composite_std": 0.3671012222766876, "reward_total_mean": 0.43106603622436523, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.43106603622436523, "rewards/meter/std": 0.3671012222766876, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.43106603622436523, "rewards/total_composite/std": 0.3671012222766876, "sampling/importance_sampling_ratio/max": 1.4617520570755005, "sampling/importance_sampling_ratio/mean": 1.0158692598342896, "sampling/importance_sampling_ratio/min": 0.2837205231189728, "sampling/sampling_logp_difference/max": 1.259765625, "sampling/sampling_logp_difference/mean": 0.059371624141931534, "step": 1430 }, { "clip_ratio/high_max": 0.027468874352052808, "clip_ratio/high_mean": 0.027468874352052808, "clip_ratio/low_mean": 0.008522727526724339, "clip_ratio/low_min": 0.008522727526724339, "clip_ratio/region_mean": 0.035991601878777146, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 90.75, "completions/mean_terminated_length": 90.75, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.12930846586823463, "epoch": 0.05747680443426919, "frac_reward_zero_std": 0.0, "grad_norm": 3.8664066791534424, "learning_rate": 5.666666666666667e-06, "loss": -0.0112, "num_tokens": 3220585.0, "reward": 0.9018567800521851, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9018567800521851, "reward_meter_std": 0.26490822434425354, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2649082541465759, "reward_total_composite_mean": 0.9018567800521851, "reward_total_composite_std": 0.26490822434425354, "reward_total_mean": 0.9018567800521851, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9018567800521851, "rewards/meter/std": 0.26490822434425354, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9018567800521851, "rewards/total_composite/std": 0.26490822434425354, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0014922618865967, "sampling/importance_sampling_ratio/min": 0.1888420283794403, "sampling/sampling_logp_difference/max": 1.666844367980957, "sampling/sampling_logp_difference/mean": 0.042072996497154236, "step": 1431 }, { "clip_ratio/high_max": 0.007326977560296655, "clip_ratio/high_mean": 0.007326977560296655, "clip_ratio/low_mean": 0.0031601624796167016, "clip_ratio/low_min": 0.0031601624796167016, "clip_ratio/region_mean": 0.010487140039913356, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 118.875, "completions/mean_terminated_length": 118.875, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "entropy": 0.1475573442876339, "epoch": 0.05751696991605414, "frac_reward_zero_std": 0.0, "grad_norm": 3.0692851543426514, "learning_rate": 5.663636363636364e-06, "loss": 0.002, "num_tokens": 3222928.0, "reward": 0.7309507131576538, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983639717102051, "reward_meter_std": 0.00039870149339549243, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.11921755969524384, "reward_std": 0.11907363682985306, "reward_total_composite_mean": 0.7309507131576538, "reward_total_composite_std": 0.11907364428043365, "reward_total_mean": 0.7309507131576538, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983639717102051, "rewards/meter/std": 0.00039870149339549243, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.11921755969524384, "rewards/total_composite/mean": 0.7309507131576538, "rewards/total_composite/std": 0.11907364428043365, "sampling/importance_sampling_ratio/max": 1.6000510454177856, "sampling/importance_sampling_ratio/mean": 1.0022363662719727, "sampling/importance_sampling_ratio/min": 0.1574249118566513, "sampling/sampling_logp_difference/max": 1.848806619644165, "sampling/sampling_logp_difference/mean": 0.022948887199163437, "step": 1432 }, { "clip_ratio/high_max": 0.017400611890479922, "clip_ratio/high_mean": 0.017400611890479922, "clip_ratio/low_mean": 0.01948191737756133, "clip_ratio/low_min": 0.01948191737756133, "clip_ratio/region_mean": 0.03688252926804125, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 64.25, "completions/mean_terminated_length": 64.25, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.3227595426142216, "epoch": 0.0575571353978391, "frac_reward_zero_std": 0.0, "grad_norm": 4.21342658996582, "learning_rate": 5.6606060606060606e-06, "loss": -0.0086, "num_tokens": 3224674.0, "reward": 0.7867549657821655, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7867549657821655, "reward_meter_std": 0.2720617949962616, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2720617651939392, "reward_total_composite_mean": 0.7867549657821655, "reward_total_composite_std": 0.2720617949962616, "reward_total_mean": 0.7867549657821655, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7867549657821655, "rewards/meter/std": 0.2720617949962616, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7867549657821655, "rewards/total_composite/std": 0.2720617949962616, "sampling/importance_sampling_ratio/max": 1.7156156301498413, "sampling/importance_sampling_ratio/mean": 1.0058132410049438, "sampling/importance_sampling_ratio/min": 0.19917964935302734, "sampling/sampling_logp_difference/max": 1.6135480403900146, "sampling/sampling_logp_difference/mean": 0.045874159783124924, "step": 1433 }, { "clip_ratio/high_max": 0.0194749265210703, "clip_ratio/high_mean": 0.0194749265210703, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.02178974135313183, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 63.125, "completions/mean_terminated_length": 63.125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.12332060746848583, "epoch": 0.05759730087962405, "frac_reward_zero_std": 0.0, "grad_norm": 7.206746578216553, "learning_rate": 5.657575757575759e-06, "loss": -0.0555, "num_tokens": 3226467.0, "reward": 0.9712523221969604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9712523221969604, "reward_meter_std": 0.06995871663093567, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06995872408151627, "reward_total_composite_mean": 0.9712523221969604, "reward_total_composite_std": 0.06995871663093567, "reward_total_mean": 0.9712523221969604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9712523221969604, "rewards/meter/std": 0.06995871663093567, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9712523221969604, "rewards/total_composite/std": 0.06995871663093567, "sampling/importance_sampling_ratio/max": 1.8253268003463745, "sampling/importance_sampling_ratio/mean": 1.0004560947418213, "sampling/importance_sampling_ratio/min": 0.25380071997642517, "sampling/sampling_logp_difference/max": 1.3712058067321777, "sampling/sampling_logp_difference/mean": 0.024187438189983368, "step": 1434 }, { "clip_ratio/high_max": 0.019032032461836934, "clip_ratio/high_mean": 0.019032032461836934, "clip_ratio/low_mean": 0.00872093066573143, "clip_ratio/low_min": 0.00872093066573143, "clip_ratio/region_mean": 0.027752963127568364, "completions/clipped_ratio": 0.0, "completions/max_length": 48.0, "completions/max_terminated_length": 48.0, "completions/mean_length": 45.5, "completions/mean_terminated_length": 45.5, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.1447599595412612, "epoch": 0.057637466361409005, "frac_reward_zero_std": 0.0, "grad_norm": 6.715077877044678, "learning_rate": 5.654545454545455e-06, "loss": -0.0, "num_tokens": 3228023.0, "reward": 0.8577103614807129, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8577103614807129, "reward_meter_std": 0.20525795221328735, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.20525793731212616, "reward_total_composite_mean": 0.8577103614807129, "reward_total_composite_std": 0.20525795221328735, "reward_total_mean": 0.8577103614807129, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8577103614807129, "rewards/meter/std": 0.20525795221328735, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8577103614807129, "rewards/total_composite/std": 0.20525795221328735, "sampling/importance_sampling_ratio/max": 1.5408525466918945, "sampling/importance_sampling_ratio/mean": 0.9992294311523438, "sampling/importance_sampling_ratio/min": 0.5400460958480835, "sampling/sampling_logp_difference/max": 0.6161007881164551, "sampling/sampling_logp_difference/mean": 0.021754316985607147, "step": 1435 }, { "clip_ratio/high_max": 0.024871644098311663, "clip_ratio/high_mean": 0.024871644098311663, "clip_ratio/low_mean": 0.008230874547734857, "clip_ratio/low_min": 0.008230874547734857, "clip_ratio/region_mean": 0.03310251864604652, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.13815412484109402, "epoch": 0.05767763184319396, "frac_reward_zero_std": 0.0, "grad_norm": 5.0643205642700195, "learning_rate": 5.651515151515152e-06, "loss": -0.0268, "num_tokens": 3229879.0, "reward": 0.9666499495506287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9666499495506287, "reward_meter_std": 0.03790951892733574, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03790954127907753, "reward_total_composite_mean": 0.9666499495506287, "reward_total_composite_std": 0.03790951892733574, "reward_total_mean": 0.9666499495506287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9666499495506287, "rewards/meter/std": 0.03790951892733574, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9666499495506287, "rewards/total_composite/std": 0.03790951892733574, "sampling/importance_sampling_ratio/max": 1.844899296760559, "sampling/importance_sampling_ratio/mean": 1.0077619552612305, "sampling/importance_sampling_ratio/min": 0.2260764092206955, "sampling/sampling_logp_difference/max": 1.486882209777832, "sampling/sampling_logp_difference/mean": 0.030510857701301575, "step": 1436 }, { "clip_ratio/high_max": 0.005740093300119042, "clip_ratio/high_mean": 0.005740093300119042, "clip_ratio/low_mean": 0.005988386110402644, "clip_ratio/low_min": 0.005988386110402644, "clip_ratio/region_mean": 0.011728479410521686, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 64.25, "completions/mean_terminated_length": 64.25, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.1368792960420251, "epoch": 0.05771779732497891, "frac_reward_zero_std": 0.0, "grad_norm": 7.082005500793457, "learning_rate": 5.648484848484849e-06, "loss": -0.0032, "num_tokens": 3231833.0, "reward": 0.990328311920166, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.990328311920166, "reward_meter_std": 0.005682698916643858, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005682693794369698, "reward_total_composite_mean": 0.990328311920166, "reward_total_composite_std": 0.005682698916643858, "reward_total_mean": 0.990328311920166, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.990328311920166, "rewards/meter/std": 0.005682698916643858, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.990328311920166, "rewards/total_composite/std": 0.005682698916643858, "sampling/importance_sampling_ratio/max": 1.7731237411499023, "sampling/importance_sampling_ratio/mean": 0.9941160082817078, "sampling/importance_sampling_ratio/min": 0.11656776815652847, "sampling/sampling_logp_difference/max": 2.149282455444336, "sampling/sampling_logp_difference/mean": 0.0303493719547987, "step": 1437 }, { "clip_ratio/high_max": 0.018068846315145493, "clip_ratio/high_mean": 0.018068846315145493, "clip_ratio/low_mean": 0.007591875968500972, "clip_ratio/low_min": 0.007591875968500972, "clip_ratio/region_mean": 0.025660722283646464, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 47.0, "completions/min_terminated_length": 47.0, "entropy": 0.21246233023703098, "epoch": 0.05775796280676387, "frac_reward_zero_std": 0.0, "grad_norm": 4.488318920135498, "learning_rate": 5.645454545454546e-06, "loss": -0.009, "num_tokens": 3233673.0, "reward": 0.9756955504417419, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9756955504417419, "reward_meter_std": 0.017869651317596436, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01786964386701584, "reward_total_composite_mean": 0.9756955504417419, "reward_total_composite_std": 0.017869651317596436, "reward_total_mean": 0.9756955504417419, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9756955504417419, "rewards/meter/std": 0.017869651317596436, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9756955504417419, "rewards/total_composite/std": 0.017869651317596436, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008415699005127, "sampling/importance_sampling_ratio/min": 0.2599244713783264, "sampling/sampling_logp_difference/max": 1.3473641872406006, "sampling/sampling_logp_difference/mean": 0.03772404417395592, "step": 1438 }, { "clip_ratio/high_max": 0.03688650333788246, "clip_ratio/high_mean": 0.03688650333788246, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/region_mean": 0.044462261139415205, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2875422090291977, "epoch": 0.05779812828854882, "frac_reward_zero_std": 0.0, "grad_norm": 3.8302407264709473, "learning_rate": 5.642424242424242e-06, "loss": -0.0046, "num_tokens": 3235514.0, "reward": 0.9976571798324585, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976571798324585, "reward_meter_std": 0.001123868627473712, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001123874681070447, "reward_total_composite_mean": 0.9976571798324585, "reward_total_composite_std": 0.001123868627473712, "reward_total_mean": 0.9976571798324585, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976571798324585, "rewards/meter/std": 0.001123868627473712, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976571798324585, "rewards/total_composite/std": 0.001123868627473712, "sampling/importance_sampling_ratio/max": 1.605924129486084, "sampling/importance_sampling_ratio/mean": 1.0110960006713867, "sampling/importance_sampling_ratio/min": 0.3230156898498535, "sampling/sampling_logp_difference/max": 1.1300544738769531, "sampling/sampling_logp_difference/mean": 0.04735542833805084, "step": 1439 }, { "clip_ratio/high_max": 0.025198507588356733, "clip_ratio/high_mean": 0.025198507588356733, "clip_ratio/low_mean": 0.01373849913943559, "clip_ratio/low_min": 0.01373849913943559, "clip_ratio/region_mean": 0.03893700672779232, "completions/clipped_ratio": 0.0, "completions/max_length": 211.0, "completions/max_terminated_length": 211.0, "completions/mean_length": 197.25, "completions/mean_terminated_length": 197.25, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.3795258644968271, "epoch": 0.057838293770333775, "frac_reward_zero_std": 0.0, "grad_norm": 2.74971342086792, "learning_rate": 5.6393939393939405e-06, "loss": -0.0197, "num_tokens": 3238860.0, "reward": 0.641140341758728, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.05050762742757797, "reward_meter_mean": 0.9103585481643677, "reward_meter_std": 0.11414875090122223, "reward_repeat_penalty_mean": 0.8397727012634277, "reward_repeat_penalty_std": 0.1051156297326088, "reward_std": 0.11881577223539352, "reward_total_composite_mean": 0.641140341758728, "reward_total_composite_std": 0.11881579458713531, "reward_total_mean": 0.641140341758728, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.05050762742757797, "rewards/meter/mean": 0.9103585481643677, "rewards/meter/std": 0.11414875090122223, "rewards/repeat_penalty/mean": 0.8397727012634277, "rewards/repeat_penalty/std": 0.1051156297326088, "rewards/total_composite/mean": 0.641140341758728, "rewards/total_composite/std": 0.11881579458713531, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0104938745498657, "sampling/importance_sampling_ratio/min": 0.22695519030094147, "sampling/sampling_logp_difference/max": 1.4830026626586914, "sampling/sampling_logp_difference/mean": 0.052030667662620544, "step": 1440 }, { "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/low_mean": 0.009132996667176485, "clip_ratio/low_min": 0.009132996667176485, "clip_ratio/region_mean": 0.01367845106869936, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 54.75, "completions/mean_terminated_length": 54.75, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.15620220359414816, "epoch": 0.05787845925211873, "frac_reward_zero_std": 0.0, "grad_norm": 5.571752548217773, "learning_rate": 5.636363636363636e-06, "loss": -0.0023, "num_tokens": 3240498.0, "reward": 0.9891074895858765, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9891074895858765, "reward_meter_std": 0.0014680020976811647, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0014680068707093596, "reward_total_composite_mean": 0.9891074895858765, "reward_total_composite_std": 0.0014680020976811647, "reward_total_mean": 0.9891074895858765, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9891074895858765, "rewards/meter/std": 0.0014680020976811647, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9891074895858765, "rewards/total_composite/std": 0.0014680020976811647, "sampling/importance_sampling_ratio/max": 1.8307318687438965, "sampling/importance_sampling_ratio/mean": 1.0070688724517822, "sampling/importance_sampling_ratio/min": 0.41817599534988403, "sampling/sampling_logp_difference/max": 0.8718528747558594, "sampling/sampling_logp_difference/mean": 0.022303204983472824, "step": 1441 }, { "clip_ratio/high_max": 0.030609328765422106, "clip_ratio/high_mean": 0.030609328765422106, "clip_ratio/low_mean": 0.0417820424772799, "clip_ratio/low_min": 0.0417820424772799, "clip_ratio/region_mean": 0.07239137124270201, "completions/clipped_ratio": 0.0, "completions/max_length": 182.0, "completions/max_terminated_length": 182.0, "completions/mean_length": 168.5, "completions/mean_terminated_length": 168.5, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "entropy": 0.7288497872650623, "epoch": 0.05791862473390368, "frac_reward_zero_std": 0.0, "grad_norm": 4.191330909729004, "learning_rate": 5.633333333333334e-06, "loss": 0.0076, "num_tokens": 3243758.0, "reward": 0.07739339768886566, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6250000596046448, "reward_count_adherence_std": 0.05750546231865883, "reward_meter_mean": 0.12805184721946716, "reward_meter_std": 0.1377500742673874, "reward_repeat_penalty_mean": 0.9217171669006348, "reward_repeat_penalty_std": 0.1535457819700241, "reward_std": 0.08249951899051666, "reward_total_composite_mean": 0.07739339768886566, "reward_total_composite_std": 0.08249951899051666, "reward_total_mean": 0.07739339768886566, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6250000596046448, "rewards/count_adherence/std": 0.05750546231865883, "rewards/meter/mean": 0.12805184721946716, "rewards/meter/std": 0.1377500742673874, "rewards/repeat_penalty/mean": 0.9217171669006348, "rewards/repeat_penalty/std": 0.1535457819700241, "rewards/total_composite/mean": 0.07739339768886566, "rewards/total_composite/std": 0.08249951899051666, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0114877223968506, "sampling/importance_sampling_ratio/min": 0.02640198916196823, "sampling/sampling_logp_difference/max": 3.6343159675598145, "sampling/sampling_logp_difference/mean": 0.089544378221035, "step": 1442 }, { "clip_ratio/high_max": 0.026962524512782693, "clip_ratio/high_mean": 0.026962524512782693, "clip_ratio/low_mean": 0.008868969744071364, "clip_ratio/low_min": 0.008868969744071364, "clip_ratio/region_mean": 0.03583149425685406, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 274.125, "completions/mean_terminated_length": 274.125, "completions/min_length": 263.0, "completions/min_terminated_length": 263.0, "entropy": 0.34477731212973595, "epoch": 0.05795879021568864, "frac_reward_zero_std": 0.0, "grad_norm": 2.3659539222717285, "learning_rate": 5.630303030303031e-06, "loss": -0.0031, "num_tokens": 3247759.0, "reward": 0.6270208954811096, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7250000238418579, "reward_count_adherence_std": 0.04629101604223251, "reward_meter_mean": 0.993506908416748, "reward_meter_std": 0.008024541661143303, "reward_repeat_penalty_mean": 0.8730769157409668, "reward_repeat_penalty_std": 0.05622680485248566, "reward_std": 0.023548215627670288, "reward_total_composite_mean": 0.6270208954811096, "reward_total_composite_std": 0.023548215627670288, "reward_total_mean": 0.6270208954811096, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7250000238418579, "rewards/count_adherence/std": 0.04629101604223251, "rewards/meter/mean": 0.993506908416748, "rewards/meter/std": 0.008024541661143303, "rewards/repeat_penalty/mean": 0.8730769157409668, "rewards/repeat_penalty/std": 0.05622680485248566, "rewards/total_composite/mean": 0.6270208954811096, "rewards/total_composite/std": 0.023548215627670288, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0080655813217163, "sampling/importance_sampling_ratio/min": 0.0695982277393341, "sampling/sampling_logp_difference/max": 2.6650161743164062, "sampling/sampling_logp_difference/mean": 0.04566960409283638, "step": 1443 }, { "clip_ratio/high_max": 0.035727029433473945, "clip_ratio/high_mean": 0.035727029433473945, "clip_ratio/low_mean": 0.006868131808005273, "clip_ratio/low_min": 0.006868131808005273, "clip_ratio/region_mean": 0.04259516124147922, "completions/clipped_ratio": 0.0, "completions/max_length": 154.0, "completions/max_terminated_length": 154.0, "completions/mean_length": 150.25, "completions/mean_terminated_length": 150.25, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.3815644532442093, "epoch": 0.05799895569747359, "frac_reward_zero_std": 0.0, "grad_norm": 3.132058620452881, "learning_rate": 5.627272727272728e-06, "loss": -0.0009, "num_tokens": 3250457.0, "reward": 0.9605348110198975, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959889054298401, "reward_meter_std": 0.00317049166187644, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06770440936088562, "reward_total_composite_mean": 0.9605348110198975, "reward_total_composite_std": 0.06770440191030502, "reward_total_mean": 0.9605348110198975, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959889054298401, "rewards/meter/std": 0.00317049166187644, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9605348110198975, "rewards/total_composite/std": 0.06770440191030502, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0069429874420166, "sampling/importance_sampling_ratio/min": 0.23410192131996155, "sampling/sampling_logp_difference/max": 1.4519987106323242, "sampling/sampling_logp_difference/mean": 0.05538027361035347, "step": 1444 }, { "clip_ratio/high_max": 0.02173018571920693, "clip_ratio/high_mean": 0.02173018571920693, "clip_ratio/low_mean": 0.012751690810546279, "clip_ratio/low_min": 0.012751690810546279, "clip_ratio/region_mean": 0.03448187652975321, "completions/clipped_ratio": 0.0, "completions/max_length": 196.0, "completions/max_terminated_length": 196.0, "completions/mean_length": 188.375, "completions/mean_terminated_length": 188.375, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "entropy": 0.38519781827926636, "epoch": 0.058039121179258545, "frac_reward_zero_std": 0.0, "grad_norm": 3.3944966793060303, "learning_rate": 5.624242424242424e-06, "loss": 0.0052, "num_tokens": 3253588.0, "reward": 0.9414394497871399, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9968078136444092, "reward_meter_std": 0.0010806269710883498, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.08399210125207901, "reward_std": 0.08388549834489822, "reward_total_composite_mean": 0.9414394497871399, "reward_total_composite_std": 0.08388549834489822, "reward_total_mean": 0.9414394497871399, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9968078136444092, "rewards/meter/std": 0.0010806269710883498, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.9414394497871399, "rewards/total_composite/std": 0.08388549834489822, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.013535737991333, "sampling/importance_sampling_ratio/min": 0.18771031498908997, "sampling/sampling_logp_difference/max": 1.6728553771972656, "sampling/sampling_logp_difference/mean": 0.050331130623817444, "step": 1445 }, { "clip_ratio/high_max": 0.05117733031511307, "clip_ratio/high_mean": 0.05117733031511307, "clip_ratio/low_mean": 0.017605633474886417, "clip_ratio/low_min": 0.017605633474886417, "clip_ratio/region_mean": 0.06878296378999949, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 70.75, "completions/mean_terminated_length": 70.75, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.48965371400117874, "epoch": 0.0580792866610435, "frac_reward_zero_std": 0.0, "grad_norm": 5.573071479797363, "learning_rate": 5.6212121212121215e-06, "loss": 0.0122, "num_tokens": 3255370.0, "reward": 0.994280993938446, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994280993938446, "reward_meter_std": 0.0039334069006145, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003933423198759556, "reward_total_composite_mean": 0.994280993938446, "reward_total_composite_std": 0.0039334069006145, "reward_total_mean": 0.994280993938446, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994280993938446, "rewards/meter/std": 0.0039334069006145, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994280993938446, "rewards/total_composite/std": 0.0039334069006145, "sampling/importance_sampling_ratio/max": 1.8624515533447266, "sampling/importance_sampling_ratio/mean": 1.0078636407852173, "sampling/importance_sampling_ratio/min": 0.2072431445121765, "sampling/sampling_logp_difference/max": 1.5738625526428223, "sampling/sampling_logp_difference/mean": 0.07322768121957779, "step": 1446 }, { "clip_ratio/high_max": 0.03094316739588976, "clip_ratio/high_mean": 0.03094316739588976, "clip_ratio/low_mean": 0.0173611119389534, "clip_ratio/low_min": 0.0173611119389534, "clip_ratio/region_mean": 0.04830427933484316, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 35.25, "completions/mean_terminated_length": 35.25, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.27801981568336487, "epoch": 0.05811945214282845, "frac_reward_zero_std": 0.0, "grad_norm": 7.160624980926514, "learning_rate": 5.618181818181818e-06, "loss": 0.0006, "num_tokens": 3256876.0, "reward": 0.9780558943748474, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9780558943748474, "reward_meter_std": 0.022425083443522453, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.022425075992941856, "reward_total_composite_mean": 0.9780558943748474, "reward_total_composite_std": 0.022425083443522453, "reward_total_mean": 0.9780558943748474, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9780558943748474, "rewards/meter/std": 0.022425083443522453, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9780558943748474, "rewards/total_composite/std": 0.022425083443522453, "sampling/importance_sampling_ratio/max": 1.6601144075393677, "sampling/importance_sampling_ratio/mean": 1.0069624185562134, "sampling/importance_sampling_ratio/min": 0.4137475788593292, "sampling/sampling_logp_difference/max": 0.8824992179870605, "sampling/sampling_logp_difference/mean": 0.04081781581044197, "step": 1447 }, { "clip_ratio/high_max": 0.04604411777108908, "clip_ratio/high_mean": 0.04604411777108908, "clip_ratio/low_mean": 0.0032051282469183207, "clip_ratio/low_min": 0.0032051282469183207, "clip_ratio/region_mean": 0.0492492460180074, "completions/clipped_ratio": 0.0, "completions/max_length": 117.0, "completions/max_terminated_length": 117.0, "completions/mean_length": 114.0, "completions/mean_terminated_length": 114.0, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.348561218008399, "epoch": 0.058159617624613406, "frac_reward_zero_std": 0.0, "grad_norm": 2.8199713230133057, "learning_rate": 5.615151515151516e-06, "loss": 0.0162, "num_tokens": 3259212.0, "reward": 0.9709444046020508, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958944320678711, "reward_meter_std": 0.0037708207964897156, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06981503963470459, "reward_total_composite_mean": 0.9709444046020508, "reward_total_composite_std": 0.06981504708528519, "reward_total_mean": 0.9709444046020508, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958944320678711, "rewards/meter/std": 0.0037708207964897156, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9709444046020508, "rewards/total_composite/std": 0.06981504708528519, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023696422576904, "sampling/importance_sampling_ratio/min": 0.17166204750537872, "sampling/sampling_logp_difference/max": 1.7622275352478027, "sampling/sampling_logp_difference/mean": 0.05924813821911812, "step": 1448 }, { "clip_ratio/high_max": 0.011789240641519427, "clip_ratio/high_mean": 0.011789240641519427, "clip_ratio/low_mean": 0.007986687181983143, "clip_ratio/low_min": 0.007986687181983143, "clip_ratio/region_mean": 0.01977592782350257, "completions/clipped_ratio": 0.0, "completions/max_length": 178.0, "completions/max_terminated_length": 178.0, "completions/mean_length": 171.0, "completions/mean_terminated_length": 171.0, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.16733690444380045, "epoch": 0.05819978310639836, "frac_reward_zero_std": 0.0, "grad_norm": 2.3722407817840576, "learning_rate": 5.612121212121212e-06, "loss": 0.0176, "num_tokens": 3262076.0, "reward": 0.29604610800743103, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4469248056411743, "reward_meter_std": 0.35773399472236633, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.07120776921510696, "reward_std": 0.23469018936157227, "reward_total_composite_mean": 0.29604610800743103, "reward_total_composite_std": 0.23469020426273346, "reward_total_mean": 0.29604610800743103, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4469248056411743, "rewards/meter/std": 0.35773399472236633, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.07120776921510696, "rewards/total_composite/mean": 0.29604610800743103, "rewards/total_composite/std": 0.23469020426273346, "sampling/importance_sampling_ratio/max": 1.7698057889938354, "sampling/importance_sampling_ratio/mean": 1.0035395622253418, "sampling/importance_sampling_ratio/min": 0.09746988862752914, "sampling/sampling_logp_difference/max": 2.328211784362793, "sampling/sampling_logp_difference/mean": 0.025468070060014725, "step": 1449 }, { "clip_ratio/high_max": 0.015601763967424631, "clip_ratio/high_mean": 0.015601763967424631, "clip_ratio/low_mean": 0.017584156128577888, "clip_ratio/low_min": 0.017584156128577888, "clip_ratio/region_mean": 0.03318592009600252, "completions/clipped_ratio": 0.0, "completions/max_length": 197.0, "completions/max_terminated_length": 197.0, "completions/mean_length": 192.5, "completions/mean_terminated_length": 192.5, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.28528641909360886, "epoch": 0.058239948588183314, "frac_reward_zero_std": 0.0, "grad_norm": 2.1575472354888916, "learning_rate": 5.60909090909091e-06, "loss": 0.0088, "num_tokens": 3265160.0, "reward": 0.911919116973877, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949524998664856, "reward_meter_std": 0.0050539844669401646, "reward_repeat_penalty_mean": 0.9166666269302368, "reward_repeat_penalty_std": 0.07856741547584534, "reward_std": 0.07672726362943649, "reward_total_composite_mean": 0.911919116973877, "reward_total_composite_std": 0.0767272487282753, "reward_total_mean": 0.911919116973877, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949524998664856, "rewards/meter/std": 0.0050539844669401646, "rewards/repeat_penalty/mean": 0.9166666269302368, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.911919116973877, "rewards/total_composite/std": 0.0767272487282753, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0060464143753052, "sampling/importance_sampling_ratio/min": 0.11565891653299332, "sampling/sampling_logp_difference/max": 2.1571097373962402, "sampling/sampling_logp_difference/mean": 0.04098372161388397, "step": 1450 }, { "epoch": 0.058239948588183314, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 283.0769230769231, "eval_completions/max_terminated_length": 283.0769230769231, "eval_completions/mean_length": 160.5, "eval_completions/mean_terminated_length": 160.5, "eval_completions/min_length": 60.46153846153846, "eval_completions/min_terminated_length": 60.46153846153846, "eval_entropy": 0.26288481056690216, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 3265160.0, "eval_reward": 0.4260200307919429, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.8191676002282363, "eval_reward_count_adherence_std": 0.17260103730055001, "eval_reward_meter_mean": 0.5874437128122036, "eval_reward_meter_std": 0.4289410481086144, "eval_reward_repeat_penalty_mean": 0.8410639992127051, "eval_reward_repeat_penalty_std": 0.16562061241039863, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.4260200307919429, "eval_reward_total_composite_std": 0.3662373045316109, "eval_reward_total_mean": 0.4260200307919429, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.8191676002282363, "eval_rewards/count_adherence/std": 0.17260103730055001, "eval_rewards/meter/mean": 0.5874437128122036, "eval_rewards/meter/std": 0.4289410481086144, "eval_rewards/repeat_penalty/mean": 0.8410639992127051, "eval_rewards/repeat_penalty/std": 0.16562061241039863, "eval_rewards/total_composite/mean": 0.4260200307919429, "eval_rewards/total_composite/std": 0.3662373045316109, "eval_runtime": 55.2927, "eval_samples_per_second": 1.881, "eval_sampling/importance_sampling_ratio/max": 1.4355276914743276, "eval_sampling/importance_sampling_ratio/mean": 1.0070600234545195, "eval_sampling/importance_sampling_ratio/min": 0.3921516973238725, "eval_sampling/sampling_logp_difference/max": 0.9593030489408053, "eval_sampling/sampling_logp_difference/mean": 0.02707064416832649, "eval_steps_per_second": 0.235, "step": 1450 }, { "clip_ratio/high_max": 0.04324009292759001, "clip_ratio/high_mean": 0.04324009292759001, "clip_ratio/low_mean": 0.04481779085472226, "clip_ratio/low_min": 0.04481779085472226, "clip_ratio/region_mean": 0.08805788378231227, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 61.875, "completions/mean_terminated_length": 61.875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 1.44280257076025, "epoch": 0.05828011406996827, "frac_reward_zero_std": 0.0, "grad_norm": 9.900911331176758, "learning_rate": 5.606060606060606e-06, "loss": -0.0183, "num_tokens": 3266919.0, "reward": 0.5548748970031738, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5548748970031738, "reward_meter_std": 0.3548371493816376, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3548371493816376, "reward_total_composite_mean": 0.5548748970031738, "reward_total_composite_std": 0.3548371493816376, "reward_total_mean": 0.5548748970031738, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5548748970031738, "rewards/meter/std": 0.3548371493816376, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5548748970031738, "rewards/total_composite/std": 0.3548371493816376, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0315823554992676, "sampling/importance_sampling_ratio/min": 0.27214980125427246, "sampling/sampling_logp_difference/max": 1.3014025688171387, "sampling/sampling_logp_difference/mean": 0.12563803791999817, "step": 1451 }, { "clip_ratio/high_max": 0.041309412801638246, "clip_ratio/high_mean": 0.041309412801638246, "clip_ratio/low_mean": 0.013940956210717559, "clip_ratio/low_min": 0.013940956210717559, "clip_ratio/region_mean": 0.055250369012355804, "completions/clipped_ratio": 0.0, "completions/max_length": 110.0, "completions/max_terminated_length": 110.0, "completions/mean_length": 102.5, "completions/mean_terminated_length": 102.5, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.5281493403017521, "epoch": 0.05832027955175322, "frac_reward_zero_std": 0.0, "grad_norm": 3.927211046218872, "learning_rate": 5.603030303030303e-06, "loss": 0.0264, "num_tokens": 3269067.0, "reward": 0.9900820255279541, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9900820255279541, "reward_meter_std": 0.011413360014557838, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011413362808525562, "reward_total_composite_mean": 0.9900820255279541, "reward_total_composite_std": 0.011413360014557838, "reward_total_mean": 0.9900820255279541, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9900820255279541, "rewards/meter/std": 0.011413360014557838, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9900820255279541, "rewards/total_composite/std": 0.011413360014557838, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0114924907684326, "sampling/importance_sampling_ratio/min": 0.19062404334545135, "sampling/sampling_logp_difference/max": 1.65745210647583, "sampling/sampling_logp_difference/mean": 0.06290001422166824, "step": 1452 }, { "clip_ratio/high_max": 0.01592310261912644, "clip_ratio/high_mean": 0.01592310261912644, "clip_ratio/low_mean": 0.024039480136707425, "clip_ratio/low_min": 0.024039480136707425, "clip_ratio/region_mean": 0.039962582755833864, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 61.625, "completions/mean_terminated_length": 61.625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.5127889458090067, "epoch": 0.058360445033538176, "frac_reward_zero_std": 0.0, "grad_norm": 4.87343168258667, "learning_rate": 5.600000000000001e-06, "loss": 0.029, "num_tokens": 3271008.0, "reward": 0.18945728242397308, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.18945728242397308, "reward_meter_std": 0.2888922691345215, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.28889229893684387, "reward_total_composite_mean": 0.18945728242397308, "reward_total_composite_std": 0.2888922691345215, "reward_total_mean": 0.18945728242397308, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.18945728242397308, "rewards/meter/std": 0.2888922691345215, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.18945728242397308, "rewards/total_composite/std": 0.2888922691345215, "sampling/importance_sampling_ratio/max": 1.7248963117599487, "sampling/importance_sampling_ratio/mean": 1.011013150215149, "sampling/importance_sampling_ratio/min": 0.11022898554801941, "sampling/sampling_logp_difference/max": 2.205195426940918, "sampling/sampling_logp_difference/mean": 0.06080813705921173, "step": 1453 }, { "clip_ratio/high_max": 0.017170886043459177, "clip_ratio/high_mean": 0.017170886043459177, "clip_ratio/low_mean": 0.004204352619126439, "clip_ratio/low_min": 0.004204352619126439, "clip_ratio/region_mean": 0.021375238662585616, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 58.625, "completions/mean_terminated_length": 58.625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.15899417083710432, "epoch": 0.05840061051532313, "frac_reward_zero_std": 0.0, "grad_norm": 3.9825429916381836, "learning_rate": 5.596969696969697e-06, "loss": 0.0102, "num_tokens": 3272821.0, "reward": 0.9886520504951477, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9886520504951477, "reward_meter_std": 0.006697438657283783, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00669743912294507, "reward_total_composite_mean": 0.9886520504951477, "reward_total_composite_std": 0.006697438657283783, "reward_total_mean": 0.9886520504951477, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9886520504951477, "rewards/meter/std": 0.006697438657283783, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9886520504951477, "rewards/total_composite/std": 0.006697438657283783, "sampling/importance_sampling_ratio/max": 1.4476277828216553, "sampling/importance_sampling_ratio/mean": 0.9944877028465271, "sampling/importance_sampling_ratio/min": 0.33113205432891846, "sampling/sampling_logp_difference/max": 1.1052379608154297, "sampling/sampling_logp_difference/mean": 0.03104584477841854, "step": 1454 }, { "clip_ratio/high_max": 0.0423718374222517, "clip_ratio/high_mean": 0.0423718374222517, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/region_mean": 0.045844059670343995, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 36.0, "completions/mean_terminated_length": 36.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.32455965876579285, "epoch": 0.058440775997108084, "frac_reward_zero_std": 0.0, "grad_norm": 7.097043514251709, "learning_rate": 5.593939393939395e-06, "loss": 0.0227, "num_tokens": 3274349.0, "reward": 0.9973074793815613, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973074793815613, "reward_meter_std": 0.0026173454243689775, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0026173454243689775, "reward_total_composite_mean": 0.9973074793815613, "reward_total_composite_std": 0.0026173454243689775, "reward_total_mean": 0.9973074793815613, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973074793815613, "rewards/meter/std": 0.0026173454243689775, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973074793815613, "rewards/total_composite/std": 0.0026173454243689775, "sampling/importance_sampling_ratio/max": 1.5138667821884155, "sampling/importance_sampling_ratio/mean": 0.9993876218795776, "sampling/importance_sampling_ratio/min": 0.5086050033569336, "sampling/sampling_logp_difference/max": 0.6760835647583008, "sampling/sampling_logp_difference/mean": 0.04888688772916794, "step": 1455 }, { "clip_ratio/high_max": 0.03280176408588886, "clip_ratio/high_mean": 0.03280176408588886, "clip_ratio/low_mean": 0.012334733735769987, "clip_ratio/low_min": 0.012334733735769987, "clip_ratio/region_mean": 0.04513649782165885, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 263.25, "completions/mean_terminated_length": 263.25, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.41418151557445526, "epoch": 0.05848094147889304, "frac_reward_zero_std": 0.0, "grad_norm": 2.318981170654297, "learning_rate": 5.5909090909090915e-06, "loss": -0.0485, "num_tokens": 3278247.0, "reward": 0.7669380903244019, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.9962884187698364, "reward_meter_std": 0.0015494668623432517, "reward_repeat_penalty_mean": 0.911057710647583, "reward_repeat_penalty_std": 0.051787521690130234, "reward_std": 0.07985157519578934, "reward_total_composite_mean": 0.7669380903244019, "reward_total_composite_std": 0.07985159009695053, "reward_total_mean": 0.7669380903244019, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.9962884187698364, "rewards/meter/std": 0.0015494668623432517, "rewards/repeat_penalty/mean": 0.911057710647583, "rewards/repeat_penalty/std": 0.051787521690130234, "rewards/total_composite/mean": 0.7669380903244019, "rewards/total_composite/std": 0.07985159009695053, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0123025178909302, "sampling/importance_sampling_ratio/min": 0.2368977963924408, "sampling/sampling_logp_difference/max": 1.4401264190673828, "sampling/sampling_logp_difference/mean": 0.05095827206969261, "step": 1456 }, { "clip_ratio/high_max": 0.04439025931060314, "clip_ratio/high_mean": 0.04439025931060314, "clip_ratio/low_mean": 0.00958982715383172, "clip_ratio/low_min": 0.00958982715383172, "clip_ratio/region_mean": 0.05398008646443486, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 96.75, "completions/mean_terminated_length": 96.75, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.3380627855658531, "epoch": 0.05852110696067799, "frac_reward_zero_std": 0.0, "grad_norm": 2.4325199127197266, "learning_rate": 5.587878787878789e-06, "loss": -0.0297, "num_tokens": 3280349.0, "reward": 0.8396018743515015, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9892526268959045, "reward_meter_std": 0.015538093633949757, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.2070196568965912, "reward_std": 0.20004859566688538, "reward_total_composite_mean": 0.8396018743515015, "reward_total_composite_std": 0.20004859566688538, "reward_total_mean": 0.8396018743515015, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9892526268959045, "rewards/meter/std": 0.015538093633949757, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.2070196568965912, "rewards/total_composite/mean": 0.8396018743515015, "rewards/total_composite/std": 0.20004859566688538, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023468732833862, "sampling/importance_sampling_ratio/min": 0.13196192681789398, "sampling/sampling_logp_difference/max": 2.0252418518066406, "sampling/sampling_logp_difference/mean": 0.05370476841926575, "step": 1457 }, { "clip_ratio/high_max": 0.022278774995356798, "clip_ratio/high_mean": 0.022278774995356798, "clip_ratio/low_mean": 0.009437821572646499, "clip_ratio/low_min": 0.009437821572646499, "clip_ratio/region_mean": 0.0317165965680033, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 93.5, "completions/mean_terminated_length": 93.5, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.24885553307831287, "epoch": 0.058561272442462946, "frac_reward_zero_std": 0.0, "grad_norm": 4.565279006958008, "learning_rate": 5.584848484848485e-06, "loss": -0.0117, "num_tokens": 3282489.0, "reward": 0.772584080696106, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970370531082153, "reward_meter_std": 0.0014831003500148654, "reward_repeat_penalty_mean": 0.7750000357627869, "reward_repeat_penalty_std": 0.16690459847450256, "reward_std": 0.1656908094882965, "reward_total_composite_mean": 0.772584080696106, "reward_total_composite_std": 0.1656908094882965, "reward_total_mean": 0.772584080696106, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970370531082153, "rewards/meter/std": 0.0014831003500148654, "rewards/repeat_penalty/mean": 0.7750000357627869, "rewards/repeat_penalty/std": 0.16690459847450256, "rewards/total_composite/mean": 0.772584080696106, "rewards/total_composite/std": 0.1656908094882965, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0047050714492798, "sampling/importance_sampling_ratio/min": 0.31182003021240234, "sampling/sampling_logp_difference/max": 1.165329098701477, "sampling/sampling_logp_difference/mean": 0.03978914022445679, "step": 1458 }, { "clip_ratio/high_max": 0.03423361713066697, "clip_ratio/high_mean": 0.03423361713066697, "clip_ratio/low_mean": 0.014779910678043962, "clip_ratio/low_min": 0.014779910678043962, "clip_ratio/region_mean": 0.04901352780871093, "completions/clipped_ratio": 0.0, "completions/max_length": 258.0, "completions/max_terminated_length": 258.0, "completions/mean_length": 223.75, "completions/mean_terminated_length": 223.75, "completions/min_length": 200.0, "completions/min_terminated_length": 200.0, "entropy": 0.5131779182702303, "epoch": 0.0586014379242479, "frac_reward_zero_std": 0.0, "grad_norm": 3.691830635070801, "learning_rate": 5.5818181818181824e-06, "loss": -0.0396, "num_tokens": 3285975.0, "reward": 0.12107732146978378, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5576923489570618, "reward_count_adherence_std": 0.05439283326268196, "reward_meter_mean": 0.2490365207195282, "reward_meter_std": 0.21091364324092865, "reward_repeat_penalty_mean": 0.8426282405853271, "reward_repeat_penalty_std": 0.1708158552646637, "reward_std": 0.09244445711374283, "reward_total_composite_mean": 0.12107732146978378, "reward_total_composite_std": 0.09244445711374283, "reward_total_mean": 0.12107732146978378, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5576923489570618, "rewards/count_adherence/std": 0.05439283326268196, "rewards/meter/mean": 0.2490365207195282, "rewards/meter/std": 0.21091364324092865, "rewards/repeat_penalty/mean": 0.8426282405853271, "rewards/repeat_penalty/std": 0.1708158552646637, "rewards/total_composite/mean": 0.12107732146978378, "rewards/total_composite/std": 0.09244445711374283, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0115958452224731, "sampling/importance_sampling_ratio/min": 0.19063332676887512, "sampling/sampling_logp_difference/max": 1.6574034690856934, "sampling/sampling_logp_difference/mean": 0.0653611496090889, "step": 1459 }, { "clip_ratio/high_max": 0.026631702319718897, "clip_ratio/high_mean": 0.026631702319718897, "clip_ratio/low_mean": 0.003759611048735678, "clip_ratio/low_min": 0.003759611048735678, "clip_ratio/region_mean": 0.030391313368454576, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.1356327636167407, "epoch": 0.058641603406032854, "frac_reward_zero_std": 0.0, "grad_norm": 4.334770679473877, "learning_rate": 5.578787878787879e-06, "loss": 0.0048, "num_tokens": 3287646.0, "reward": 0.9952852129936218, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952852129936218, "reward_meter_std": 0.0014361762441694736, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0014361703069880605, "reward_total_composite_mean": 0.9952852129936218, "reward_total_composite_std": 0.0014361762441694736, "reward_total_mean": 0.9952852129936218, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952852129936218, "rewards/meter/std": 0.0014361762441694736, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952852129936218, "rewards/total_composite/std": 0.0014361762441694736, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0027287006378174, "sampling/importance_sampling_ratio/min": 0.24768902361392975, "sampling/sampling_logp_difference/max": 1.3955812454223633, "sampling/sampling_logp_difference/mean": 0.03258511424064636, "step": 1460 }, { "clip_ratio/high_max": 0.016740256920456886, "clip_ratio/high_mean": 0.016740256920456886, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/region_mean": 0.02507359068840742, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.24057043716311455, "epoch": 0.05868176888781781, "frac_reward_zero_std": 0.0, "grad_norm": 7.510387420654297, "learning_rate": 5.575757575757577e-06, "loss": 0.0375, "num_tokens": 3289631.0, "reward": 0.8637334108352661, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8637334108352661, "reward_meter_std": 0.34701576828956604, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.34701573848724365, "reward_total_composite_mean": 0.8637334108352661, "reward_total_composite_std": 0.34701576828956604, "reward_total_mean": 0.8637334108352661, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8637334108352661, "rewards/meter/std": 0.34701576828956604, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8637334108352661, "rewards/total_composite/std": 0.34701576828956604, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0041791200637817, "sampling/importance_sampling_ratio/min": 0.3521841764450073, "sampling/sampling_logp_difference/max": 1.0436010360717773, "sampling/sampling_logp_difference/mean": 0.03330973535776138, "step": 1461 }, { "clip_ratio/high_max": 0.02201478136703372, "clip_ratio/high_mean": 0.02201478136703372, "clip_ratio/low_mean": 0.029018934816122055, "clip_ratio/low_min": 0.029018934816122055, "clip_ratio/region_mean": 0.051033716183155775, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 69.5, "completions/mean_terminated_length": 69.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.5490291193127632, "epoch": 0.05872193436960276, "frac_reward_zero_std": 0.0, "grad_norm": 5.110687255859375, "learning_rate": 5.572727272727273e-06, "loss": 0.0109, "num_tokens": 3291531.0, "reward": 0.9955259561538696, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955259561538696, "reward_meter_std": 0.0017520119436085224, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001752000767737627, "reward_total_composite_mean": 0.9955259561538696, "reward_total_composite_std": 0.0017520119436085224, "reward_total_mean": 0.9955259561538696, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955259561538696, "rewards/meter/std": 0.0017520119436085224, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955259561538696, "rewards/total_composite/std": 0.0017520119436085224, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.018951654434204, "sampling/importance_sampling_ratio/min": 0.26090142130851746, "sampling/sampling_logp_difference/max": 1.3436126708984375, "sampling/sampling_logp_difference/mean": 0.06455224007368088, "step": 1462 }, { "clip_ratio/high_max": 0.02786910650320351, "clip_ratio/high_mean": 0.02786910650320351, "clip_ratio/low_mean": 0.027423894964158535, "clip_ratio/low_min": 0.027423894964158535, "clip_ratio/region_mean": 0.055293001467362046, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.5481097176671028, "epoch": 0.058762099851387715, "frac_reward_zero_std": 0.0, "grad_norm": 5.067295551300049, "learning_rate": 5.569696969696971e-06, "loss": -0.0018, "num_tokens": 3292987.0, "reward": 0.7616782188415527, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7616782188415527, "reward_meter_std": 0.24172598123550415, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.24172596633434296, "reward_total_composite_mean": 0.7616782188415527, "reward_total_composite_std": 0.24172598123550415, "reward_total_mean": 0.7616782188415527, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7616782188415527, "rewards/meter/std": 0.24172598123550415, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7616782188415527, "rewards/total_composite/std": 0.24172598123550415, "sampling/importance_sampling_ratio/max": 1.8223750591278076, "sampling/importance_sampling_ratio/mean": 1.0115406513214111, "sampling/importance_sampling_ratio/min": 0.2367541640996933, "sampling/sampling_logp_difference/max": 1.4407329559326172, "sampling/sampling_logp_difference/mean": 0.07128675282001495, "step": 1463 }, { "clip_ratio/high_max": 0.02262367820367217, "clip_ratio/high_mean": 0.02262367820367217, "clip_ratio/low_mean": 0.01695890142582357, "clip_ratio/low_min": 0.01695890142582357, "clip_ratio/region_mean": 0.03958257962949574, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 59.5, "completions/mean_terminated_length": 59.5, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.5435944311320782, "epoch": 0.05880226533317267, "frac_reward_zero_std": 0.0, "grad_norm": 5.747054100036621, "learning_rate": 5.566666666666667e-06, "loss": -0.0081, "num_tokens": 3294759.0, "reward": 0.33728209137916565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.33728209137916565, "reward_meter_std": 0.2937431037425995, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2937430739402771, "reward_total_composite_mean": 0.33728209137916565, "reward_total_composite_std": 0.2937431037425995, "reward_total_mean": 0.33728209137916565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.33728209137916565, "rewards/meter/std": 0.2937431037425995, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.33728209137916565, "rewards/total_composite/std": 0.2937431037425995, "sampling/importance_sampling_ratio/max": 1.9918129444122314, "sampling/importance_sampling_ratio/mean": 1.0084482431411743, "sampling/importance_sampling_ratio/min": 0.40699464082717896, "sampling/sampling_logp_difference/max": 0.8989553451538086, "sampling/sampling_logp_difference/mean": 0.06646484136581421, "step": 1464 }, { "clip_ratio/high_max": 0.02795016940217465, "clip_ratio/high_mean": 0.02795016940217465, "clip_ratio/low_mean": 0.010546227567829192, "clip_ratio/low_min": 0.010546227567829192, "clip_ratio/region_mean": 0.03849639697000384, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 67.375, "completions/mean_terminated_length": 67.375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.4691624492406845, "epoch": 0.05884243081495762, "frac_reward_zero_std": 0.0, "grad_norm": 5.774188041687012, "learning_rate": 5.563636363636364e-06, "loss": 0.0259, "num_tokens": 3296618.0, "reward": 0.9969590306282043, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969590306282043, "reward_meter_std": 0.0012544745113700628, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0012544887140393257, "reward_total_composite_mean": 0.9969590306282043, "reward_total_composite_std": 0.0012544745113700628, "reward_total_mean": 0.9969590306282043, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969590306282043, "rewards/meter/std": 0.0012544745113700628, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969590306282043, "rewards/total_composite/std": 0.0012544745113700628, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0145848989486694, "sampling/importance_sampling_ratio/min": 0.13399209082126617, "sampling/sampling_logp_difference/max": 2.009974479675293, "sampling/sampling_logp_difference/mean": 0.06404515355825424, "step": 1465 }, { "clip_ratio/high_max": 0.025272560073062778, "clip_ratio/high_mean": 0.025272560073062778, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/region_mean": 0.02735589351505041, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 59.125, "completions/mean_terminated_length": 59.125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.2567995712161064, "epoch": 0.05888259629674258, "frac_reward_zero_std": 0.0, "grad_norm": 6.224666118621826, "learning_rate": 5.560606060606061e-06, "loss": 0.0091, "num_tokens": 3298395.0, "reward": 0.9083940982818604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9083940982818604, "reward_meter_std": 0.1951943337917328, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1951943188905716, "reward_total_composite_mean": 0.9083940982818604, "reward_total_composite_std": 0.1951943337917328, "reward_total_mean": 0.9083940982818604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9083940982818604, "rewards/meter/std": 0.1951943337917328, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9083940982818604, "rewards/total_composite/std": 0.1951943337917328, "sampling/importance_sampling_ratio/max": 1.7400184869766235, "sampling/importance_sampling_ratio/mean": 1.0091822147369385, "sampling/importance_sampling_ratio/min": 0.40173083543777466, "sampling/sampling_logp_difference/max": 0.9119729995727539, "sampling/sampling_logp_difference/mean": 0.030300971120595932, "step": 1466 }, { "clip_ratio/high_max": 0.014436864759773016, "clip_ratio/high_mean": 0.014436864759773016, "clip_ratio/low_mean": 0.024785209679976106, "clip_ratio/low_min": 0.024785209679976106, "clip_ratio/region_mean": 0.03922207443974912, "completions/clipped_ratio": 0.0, "completions/max_length": 236.0, "completions/max_terminated_length": 236.0, "completions/mean_length": 220.625, "completions/mean_terminated_length": 220.625, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "entropy": 0.3614166211336851, "epoch": 0.05892276177852753, "frac_reward_zero_std": 0.0, "grad_norm": 3.2179627418518066, "learning_rate": 5.557575757575758e-06, "loss": -0.0184, "num_tokens": 3301728.0, "reward": 0.1352500319480896, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5288461446762085, "reward_count_adherence_std": 0.027196424081921577, "reward_meter_mean": 0.3030664324760437, "reward_meter_std": 0.29540300369262695, "reward_repeat_penalty_mean": 0.8133013248443604, "reward_repeat_penalty_std": 0.10961552709341049, "reward_std": 0.13207849860191345, "reward_total_composite_mean": 0.1352500319480896, "reward_total_composite_std": 0.13207849860191345, "reward_total_mean": 0.1352500319480896, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5288461446762085, "rewards/count_adherence/std": 0.027196424081921577, "rewards/meter/mean": 0.3030664324760437, "rewards/meter/std": 0.29540300369262695, "rewards/repeat_penalty/mean": 0.8133013248443604, "rewards/repeat_penalty/std": 0.10961552709341049, "rewards/total_composite/mean": 0.1352500319480896, "rewards/total_composite/std": 0.13207849860191345, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006094217300415, "sampling/importance_sampling_ratio/min": 0.13792794942855835, "sampling/sampling_logp_difference/max": 1.9810237884521484, "sampling/sampling_logp_difference/mean": 0.05384691804647446, "step": 1467 }, { "clip_ratio/high_max": 0.023276393418200314, "clip_ratio/high_mean": 0.023276393418200314, "clip_ratio/low_mean": 0.0190093262353912, "clip_ratio/low_min": 0.0190093262353912, "clip_ratio/region_mean": 0.042285719653591514, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.5, "completions/mean_terminated_length": 70.5, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.5614041350781918, "epoch": 0.058962927260312485, "frac_reward_zero_std": 0.0, "grad_norm": 3.850555419921875, "learning_rate": 5.554545454545454e-06, "loss": 0.0394, "num_tokens": 3303612.0, "reward": 0.993515133857727, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.993515133857727, "reward_meter_std": 0.002871233271434903, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0028712216299027205, "reward_total_composite_mean": 0.993515133857727, "reward_total_composite_std": 0.002871233271434903, "reward_total_mean": 0.993515133857727, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.993515133857727, "rewards/meter/std": 0.002871233271434903, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993515133857727, "rewards/total_composite/std": 0.002871233271434903, "sampling/importance_sampling_ratio/max": 1.9790068864822388, "sampling/importance_sampling_ratio/mean": 1.0134084224700928, "sampling/importance_sampling_ratio/min": 0.3374238908290863, "sampling/sampling_logp_difference/max": 1.0864152908325195, "sampling/sampling_logp_difference/mean": 0.06810689717531204, "step": 1468 }, { "clip_ratio/high_max": 0.025160838733427227, "clip_ratio/high_mean": 0.025160838733427227, "clip_ratio/low_mean": 0.0071839080192148685, "clip_ratio/low_min": 0.0071839080192148685, "clip_ratio/region_mean": 0.032344746752642095, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 92.625, "completions/mean_terminated_length": 92.625, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.2813885472714901, "epoch": 0.05900309274209744, "frac_reward_zero_std": 0.0, "grad_norm": 3.2438156604766846, "learning_rate": 5.5515151515151524e-06, "loss": -0.0341, "num_tokens": 3305697.0, "reward": 0.7041343450546265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7984445691108704, "reward_meter_std": 0.2091599404811859, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.17728103697299957, "reward_std": 0.2718327045440674, "reward_total_composite_mean": 0.7041343450546265, "reward_total_composite_std": 0.271832674741745, "reward_total_mean": 0.7041343450546265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7984445691108704, "rewards/meter/std": 0.2091599404811859, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.17728103697299957, "rewards/total_composite/mean": 0.7041343450546265, "rewards/total_composite/std": 0.271832674741745, "sampling/importance_sampling_ratio/max": 1.4045096635818481, "sampling/importance_sampling_ratio/mean": 1.0063135623931885, "sampling/importance_sampling_ratio/min": 0.2825163006782532, "sampling/sampling_logp_difference/max": 1.2640190124511719, "sampling/sampling_logp_difference/mean": 0.041127223521471024, "step": 1469 }, { "clip_ratio/high_max": 0.04982595471665263, "clip_ratio/high_mean": 0.04982595471665263, "clip_ratio/low_mean": 0.013322061393409967, "clip_ratio/low_min": 0.013322061393409967, "clip_ratio/region_mean": 0.0631480161100626, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 70.375, "completions/mean_terminated_length": 70.375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.7106457650661469, "epoch": 0.05904325822388239, "frac_reward_zero_std": 0.0, "grad_norm": 6.198017120361328, "learning_rate": 5.548484848484849e-06, "loss": -0.0266, "num_tokens": 3307476.0, "reward": 0.8553205132484436, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8553205132484436, "reward_meter_std": 0.22217348217964172, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22217348217964172, "reward_total_composite_mean": 0.8553205132484436, "reward_total_composite_std": 0.22217348217964172, "reward_total_mean": 0.8553205132484436, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8553205132484436, "rewards/meter/std": 0.22217348217964172, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8553205132484436, "rewards/total_composite/std": 0.22217348217964172, "sampling/importance_sampling_ratio/max": 1.9982537031173706, "sampling/importance_sampling_ratio/mean": 1.0185297727584839, "sampling/importance_sampling_ratio/min": 0.3050316274166107, "sampling/sampling_logp_difference/max": 1.1873397827148438, "sampling/sampling_logp_difference/mean": 0.07493096590042114, "step": 1470 }, { "clip_ratio/high_max": 0.021573547972366214, "clip_ratio/high_mean": 0.021573547972366214, "clip_ratio/low_mean": 0.011254789307713509, "clip_ratio/low_min": 0.011254789307713509, "clip_ratio/region_mean": 0.03282833728007972, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 56.75, "completions/mean_terminated_length": 56.75, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.237571744248271, "epoch": 0.05908342370566735, "frac_reward_zero_std": 0.0, "grad_norm": 4.865560054779053, "learning_rate": 5.545454545454546e-06, "loss": -0.0016, "num_tokens": 3309226.0, "reward": 0.9023252725601196, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9437645673751831, "reward_meter_std": 0.12552663683891296, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.15711423754692078, "reward_total_composite_mean": 0.9023252725601196, "reward_total_composite_std": 0.15711426734924316, "reward_total_mean": 0.9023252725601196, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9437645673751831, "rewards/meter/std": 0.12552663683891296, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9023252725601196, "rewards/total_composite/std": 0.15711426734924316, "sampling/importance_sampling_ratio/max": 1.7097090482711792, "sampling/importance_sampling_ratio/mean": 1.0094138383865356, "sampling/importance_sampling_ratio/min": 0.196671262383461, "sampling/sampling_logp_difference/max": 1.6262216567993164, "sampling/sampling_logp_difference/mean": 0.03652970865368843, "step": 1471 }, { "clip_ratio/high_max": 0.009076492046006024, "clip_ratio/high_mean": 0.009076492046006024, "clip_ratio/low_mean": 0.005906250094994903, "clip_ratio/low_min": 0.005906250094994903, "clip_ratio/region_mean": 0.014982742141000926, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 123.5, "completions/mean_terminated_length": 123.5, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.21886500529944897, "epoch": 0.0591235891874523, "frac_reward_zero_std": 0.0, "grad_norm": 2.8264143466949463, "learning_rate": 5.5424242424242425e-06, "loss": -0.0156, "num_tokens": 3311622.0, "reward": 0.5241959095001221, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7211212515830994, "reward_meter_std": 0.2165415734052658, "reward_repeat_penalty_mean": 0.7321428060531616, "reward_repeat_penalty_std": 0.1937432438135147, "reward_std": 0.20148435235023499, "reward_total_composite_mean": 0.5241959095001221, "reward_total_composite_std": 0.20148435235023499, "reward_total_mean": 0.5241959095001221, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7211212515830994, "rewards/meter/std": 0.2165415734052658, "rewards/repeat_penalty/mean": 0.7321428060531616, "rewards/repeat_penalty/std": 0.1937432438135147, "rewards/total_composite/mean": 0.5241959095001221, "rewards/total_composite/std": 0.20148435235023499, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0071460008621216, "sampling/importance_sampling_ratio/min": 0.28034695982933044, "sampling/sampling_logp_difference/max": 1.2717273235321045, "sampling/sampling_logp_difference/mean": 0.03288319706916809, "step": 1472 }, { "clip_ratio/high_max": 0.006139094999525696, "clip_ratio/high_mean": 0.006139094999525696, "clip_ratio/low_mean": 0.005232378258369863, "clip_ratio/low_min": 0.005232378258369863, "clip_ratio/region_mean": 0.011371473257895559, "completions/clipped_ratio": 0.0, "completions/max_length": 202.0, "completions/max_terminated_length": 202.0, "completions/mean_length": 184.125, "completions/mean_terminated_length": 184.125, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.12613181956112385, "epoch": 0.059163754669237255, "frac_reward_zero_std": 0.0, "grad_norm": 2.3183512687683105, "learning_rate": 5.53939393939394e-06, "loss": 0.0149, "num_tokens": 3314535.0, "reward": 0.5370750427246094, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7857142686843872, "reward_count_adherence_std": 0.07636035233736038, "reward_meter_mean": 0.8684302568435669, "reward_meter_std": 0.16792094707489014, "reward_repeat_penalty_mean": 0.7905303239822388, "reward_repeat_penalty_std": 0.09503685683012009, "reward_std": 0.12571245431900024, "reward_total_composite_mean": 0.5370750427246094, "reward_total_composite_std": 0.12571245431900024, "reward_total_mean": 0.5370750427246094, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7857142686843872, "rewards/count_adherence/std": 0.07636035233736038, "rewards/meter/mean": 0.8684302568435669, "rewards/meter/std": 0.16792094707489014, "rewards/repeat_penalty/mean": 0.7905303239822388, "rewards/repeat_penalty/std": 0.09503685683012009, "rewards/total_composite/mean": 0.5370750427246094, "rewards/total_composite/std": 0.12571245431900024, "sampling/importance_sampling_ratio/max": 1.7813695669174194, "sampling/importance_sampling_ratio/mean": 1.0009946823120117, "sampling/importance_sampling_ratio/min": 0.20334647595882416, "sampling/sampling_logp_difference/max": 1.592844009399414, "sampling/sampling_logp_difference/mean": 0.022778350859880447, "step": 1473 }, { "clip_ratio/high_max": 0.018076194741297513, "clip_ratio/high_mean": 0.018076194741297513, "clip_ratio/low_mean": 0.008916479535400867, "clip_ratio/low_min": 0.008916479535400867, "clip_ratio/region_mean": 0.02699267427669838, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 166.75, "completions/mean_terminated_length": 166.75, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.15955681633204222, "epoch": 0.05920392015102221, "frac_reward_zero_std": 0.0, "grad_norm": 2.7291924953460693, "learning_rate": 5.536363636363636e-06, "loss": 0.0044, "num_tokens": 3317453.0, "reward": 0.6380866169929504, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.959197998046875, "reward_meter_std": 0.10039316117763519, "reward_repeat_penalty_mean": 0.8055555820465088, "reward_repeat_penalty_std": 0.11501092463731766, "reward_std": 0.07123728096485138, "reward_total_composite_mean": 0.6380866169929504, "reward_total_composite_std": 0.07123729586601257, "reward_total_mean": 0.6380866169929504, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.959197998046875, "rewards/meter/std": 0.10039316117763519, "rewards/repeat_penalty/mean": 0.8055555820465088, "rewards/repeat_penalty/std": 0.11501092463731766, "rewards/total_composite/mean": 0.6380866169929504, "rewards/total_composite/std": 0.07123729586601257, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0012325048446655, "sampling/importance_sampling_ratio/min": 0.26247191429138184, "sampling/sampling_logp_difference/max": 1.337611198425293, "sampling/sampling_logp_difference/mean": 0.027119241654872894, "step": 1474 }, { "clip_ratio/high_max": 0.044174039736390114, "clip_ratio/high_mean": 0.044174039736390114, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.0477454683277756, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 102.875, "completions/mean_terminated_length": 102.875, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.4008924439549446, "epoch": 0.05924408563280716, "frac_reward_zero_std": 0.0, "grad_norm": 7.203929424285889, "learning_rate": 5.533333333333334e-06, "loss": 0.0032, "num_tokens": 3319564.0, "reward": 0.9892958402633667, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9892958402633667, "reward_meter_std": 0.02243867516517639, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.022438665851950645, "reward_total_composite_mean": 0.9892958402633667, "reward_total_composite_std": 0.02243867516517639, "reward_total_mean": 0.9892958402633667, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9892958402633667, "rewards/meter/std": 0.02243867516517639, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9892958402633667, "rewards/total_composite/std": 0.02243867516517639, "sampling/importance_sampling_ratio/max": 1.710269570350647, "sampling/importance_sampling_ratio/mean": 1.0040569305419922, "sampling/importance_sampling_ratio/min": 0.09451375156641006, "sampling/sampling_logp_difference/max": 2.3590099811553955, "sampling/sampling_logp_difference/mean": 0.06217264011502266, "step": 1475 }, { "clip_ratio/high_max": 0.015247002593241632, "clip_ratio/high_mean": 0.015247002593241632, "clip_ratio/low_mean": 0.013878762954846025, "clip_ratio/low_min": 0.013878762954846025, "clip_ratio/region_mean": 0.029125765548087656, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 120.375, "completions/mean_terminated_length": 120.375, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.45302187465131283, "epoch": 0.05928425111459212, "frac_reward_zero_std": 0.0, "grad_norm": 4.188331604003906, "learning_rate": 5.530303030303031e-06, "loss": -0.0149, "num_tokens": 3321999.0, "reward": 0.5741269588470459, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7211596369743347, "reward_meter_std": 0.29947173595428467, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.1266293078660965, "reward_std": 0.22667212784290314, "reward_total_composite_mean": 0.5741269588470459, "reward_total_composite_std": 0.22667214274406433, "reward_total_mean": 0.5741269588470459, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7211596369743347, "rewards/meter/std": 0.29947173595428467, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.1266293078660965, "rewards/total_composite/mean": 0.5741269588470459, "rewards/total_composite/std": 0.22667214274406433, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0193830728530884, "sampling/importance_sampling_ratio/min": 0.2374485731124878, "sampling/sampling_logp_difference/max": 1.7595195770263672, "sampling/sampling_logp_difference/mean": 0.05203632265329361, "step": 1476 }, { "clip_ratio/high_max": 0.055169664323329926, "clip_ratio/high_mean": 0.055169664323329926, "clip_ratio/low_mean": 0.01433747448027134, "clip_ratio/low_min": 0.01433747448027134, "clip_ratio/region_mean": 0.06950713880360126, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.25, "completions/mean_terminated_length": 68.25, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.6124710254371166, "epoch": 0.05932441659637707, "frac_reward_zero_std": 0.0, "grad_norm": 6.804332733154297, "learning_rate": 5.527272727272728e-06, "loss": 0.0153, "num_tokens": 3323777.0, "reward": 0.9923118352890015, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9923118352890015, "reward_meter_std": 0.008320425637066364, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008320420049130917, "reward_total_composite_mean": 0.9923118352890015, "reward_total_composite_std": 0.008320425637066364, "reward_total_mean": 0.9923118352890015, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9923118352890015, "rewards/meter/std": 0.008320425637066364, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9923118352890015, "rewards/total_composite/std": 0.008320425637066364, "sampling/importance_sampling_ratio/max": 1.8288415670394897, "sampling/importance_sampling_ratio/mean": 1.0044664144515991, "sampling/importance_sampling_ratio/min": 0.19280880689620972, "sampling/sampling_logp_difference/max": 1.6460561752319336, "sampling/sampling_logp_difference/mean": 0.0801224485039711, "step": 1477 }, { "clip_ratio/high_max": 0.012526939623057842, "clip_ratio/high_mean": 0.012526939623057842, "clip_ratio/low_mean": 0.012122844811528921, "clip_ratio/low_min": 0.012122844811528921, "clip_ratio/region_mean": 0.024649784434586763, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 29.75, "completions/mean_terminated_length": 29.75, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.25689505971968174, "epoch": 0.059364582078162025, "frac_reward_zero_std": 0.0, "grad_norm": 10.884235382080078, "learning_rate": 5.524242424242424e-06, "loss": 0.0228, "num_tokens": 3325231.0, "reward": 0.9929543733596802, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9929543733596802, "reward_meter_std": 0.0024007554166018963, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002400748198851943, "reward_total_composite_mean": 0.9929543733596802, "reward_total_composite_std": 0.0024007554166018963, "reward_total_mean": 0.9929543733596802, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9929543733596802, "rewards/meter/std": 0.0024007554166018963, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929543733596802, "rewards/total_composite/std": 0.0024007554166018963, "sampling/importance_sampling_ratio/max": 1.9975814819335938, "sampling/importance_sampling_ratio/mean": 1.0106008052825928, "sampling/importance_sampling_ratio/min": 0.1587689369916916, "sampling/sampling_logp_difference/max": 1.8403053283691406, "sampling/sampling_logp_difference/mean": 0.04087400063872337, "step": 1478 }, { "clip_ratio/high_max": 0.052147963899187744, "clip_ratio/high_mean": 0.052147963899187744, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/region_mean": 0.05577115237247199, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.75, "completions/mean_terminated_length": 67.75, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.6052872687578201, "epoch": 0.05940474755994698, "frac_reward_zero_std": 0.0, "grad_norm": 6.581116676330566, "learning_rate": 5.521212121212122e-06, "loss": 0.021, "num_tokens": 3327045.0, "reward": 0.8861526846885681, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8861526846885681, "reward_meter_std": 0.3106532394886017, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3106532096862793, "reward_total_composite_mean": 0.8861526846885681, "reward_total_composite_std": 0.3106532394886017, "reward_total_mean": 0.8861526846885681, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8861526846885681, "rewards/meter/std": 0.3106532394886017, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8861526846885681, "rewards/total_composite/std": 0.3106532394886017, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0218294858932495, "sampling/importance_sampling_ratio/min": 0.4099082946777344, "sampling/sampling_logp_difference/max": 0.8918218612670898, "sampling/sampling_logp_difference/mean": 0.06694039702415466, "step": 1479 }, { "clip_ratio/high_max": 0.023252843879163265, "clip_ratio/high_mean": 0.023252843879163265, "clip_ratio/low_mean": 0.027099420549347997, "clip_ratio/low_min": 0.027099420549347997, "clip_ratio/region_mean": 0.05035226442851126, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 37.0, "completions/mean_terminated_length": 37.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.5856506302952766, "epoch": 0.05944491304173193, "frac_reward_zero_std": 0.0, "grad_norm": 12.08879280090332, "learning_rate": 5.518181818181818e-06, "loss": 0.0211, "num_tokens": 3328501.0, "reward": 0.997024655342102, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997024655342102, "reward_meter_std": 0.001989408629015088, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001989412121474743, "reward_total_composite_mean": 0.997024655342102, "reward_total_composite_std": 0.001989408629015088, "reward_total_mean": 0.997024655342102, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997024655342102, "rewards/meter/std": 0.001989408629015088, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997024655342102, "rewards/total_composite/std": 0.001989408629015088, "sampling/importance_sampling_ratio/max": 1.6434557437896729, "sampling/importance_sampling_ratio/mean": 1.000410795211792, "sampling/importance_sampling_ratio/min": 0.27323752641677856, "sampling/sampling_logp_difference/max": 1.2974138259887695, "sampling/sampling_logp_difference/mean": 0.06934916973114014, "step": 1480 }, { "clip_ratio/high_max": 0.023832474602386355, "clip_ratio/high_mean": 0.023832474602386355, "clip_ratio/low_mean": 0.015805913135409355, "clip_ratio/low_min": 0.015805913135409355, "clip_ratio/region_mean": 0.03963838773779571, "completions/clipped_ratio": 0.0, "completions/max_length": 267.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 243.0, "completions/mean_terminated_length": 243.0, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "entropy": 0.5313003696501255, "epoch": 0.059485078523516886, "frac_reward_zero_std": 0.0, "grad_norm": 3.2796733379364014, "learning_rate": 5.515151515151515e-06, "loss": -0.0133, "num_tokens": 3332149.0, "reward": 0.2932712435722351, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5089285373687744, "reward_count_adherence_std": 0.025253823027014732, "reward_meter_mean": 0.7210352420806885, "reward_meter_std": 0.4204363524913788, "reward_repeat_penalty_mean": 0.848809540271759, "reward_repeat_penalty_std": 0.11776864528656006, "reward_std": 0.1669246405363083, "reward_total_composite_mean": 0.2932712435722351, "reward_total_composite_std": 0.1669246405363083, "reward_total_mean": 0.2932712435722351, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5089285373687744, "rewards/count_adherence/std": 0.025253823027014732, "rewards/meter/mean": 0.7210352420806885, "rewards/meter/std": 0.4204363524913788, "rewards/repeat_penalty/mean": 0.848809540271759, "rewards/repeat_penalty/std": 0.11776864528656006, "rewards/total_composite/mean": 0.2932712435722351, "rewards/total_composite/std": 0.1669246405363083, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0123441219329834, "sampling/importance_sampling_ratio/min": 0.1944328099489212, "sampling/sampling_logp_difference/max": 1.6376686096191406, "sampling/sampling_logp_difference/mean": 0.06317514181137085, "step": 1481 }, { "clip_ratio/high_max": 0.023007763549685478, "clip_ratio/high_mean": 0.023007763549685478, "clip_ratio/low_mean": 0.016570850741118193, "clip_ratio/low_min": 0.016570850741118193, "clip_ratio/region_mean": 0.03957861429080367, "completions/clipped_ratio": 0.0, "completions/max_length": 172.0, "completions/max_terminated_length": 172.0, "completions/mean_length": 157.75, "completions/mean_terminated_length": 157.75, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.415604081004858, "epoch": 0.05952524400530184, "frac_reward_zero_std": 0.0, "grad_norm": 4.245050430297852, "learning_rate": 5.512121212121213e-06, "loss": -0.0162, "num_tokens": 3334931.0, "reward": 0.6518079042434692, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8035714030265808, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.9921485781669617, "reward_meter_std": 0.002238625893369317, "reward_repeat_penalty_mean": 0.8219696879386902, "reward_repeat_penalty_std": 0.08952570706605911, "reward_std": 0.05643462389707565, "reward_total_composite_mean": 0.6518079042434692, "reward_total_composite_std": 0.05643462389707565, "reward_total_mean": 0.6518079042434692, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8035714030265808, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.9921485781669617, "rewards/meter/std": 0.002238625893369317, "rewards/repeat_penalty/mean": 0.8219696879386902, "rewards/repeat_penalty/std": 0.08952570706605911, "rewards/total_composite/mean": 0.6518079042434692, "rewards/total_composite/std": 0.05643462389707565, "sampling/importance_sampling_ratio/max": 1.8663012981414795, "sampling/importance_sampling_ratio/mean": 1.0066455602645874, "sampling/importance_sampling_ratio/min": 0.29460427165031433, "sampling/sampling_logp_difference/max": 1.2221221923828125, "sampling/sampling_logp_difference/mean": 0.04733636975288391, "step": 1482 }, { "clip_ratio/high_max": 0.03336897538974881, "clip_ratio/high_mean": 0.03336897538974881, "clip_ratio/low_mean": 0.014218881260603666, "clip_ratio/low_min": 0.014218881260603666, "clip_ratio/region_mean": 0.04758785665035248, "completions/clipped_ratio": 0.0, "completions/max_length": 164.0, "completions/max_terminated_length": 164.0, "completions/mean_length": 159.75, "completions/mean_terminated_length": 159.75, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "entropy": 0.6857388708740473, "epoch": 0.059565409487086794, "frac_reward_zero_std": 0.0, "grad_norm": 4.041107177734375, "learning_rate": 5.50909090909091e-06, "loss": 0.0024, "num_tokens": 3337913.0, "reward": 0.6401326656341553, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8125, "reward_count_adherence_std": 0.05892555043101311, "reward_meter_mean": 0.8998898267745972, "reward_meter_std": 0.24152173101902008, "reward_repeat_penalty_mean": 0.8854166865348816, "reward_repeat_penalty_std": 0.10751881450414658, "reward_std": 0.18946325778961182, "reward_total_composite_mean": 0.6401326656341553, "reward_total_composite_std": 0.18946325778961182, "reward_total_mean": 0.6401326656341553, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8125, "rewards/count_adherence/std": 0.05892555043101311, "rewards/meter/mean": 0.8998898267745972, "rewards/meter/std": 0.24152173101902008, "rewards/repeat_penalty/mean": 0.8854166865348816, "rewards/repeat_penalty/std": 0.10751881450414658, "rewards/total_composite/mean": 0.6401326656341553, "rewards/total_composite/std": 0.18946325778961182, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0176057815551758, "sampling/importance_sampling_ratio/min": 0.07287947088479996, "sampling/sampling_logp_difference/max": 2.618948221206665, "sampling/sampling_logp_difference/mean": 0.06710459291934967, "step": 1483 }, { "clip_ratio/high_max": 0.018196672899648547, "clip_ratio/high_mean": 0.018196672899648547, "clip_ratio/low_mean": 0.007464349502697587, "clip_ratio/low_min": 0.007464349502697587, "clip_ratio/region_mean": 0.025661022402346134, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 33.375, "completions/mean_terminated_length": 33.375, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.5462967716157436, "epoch": 0.05960557496887175, "frac_reward_zero_std": 0.0, "grad_norm": 6.229678153991699, "learning_rate": 5.506060606060607e-06, "loss": 0.0206, "num_tokens": 3339388.0, "reward": 0.9635177850723267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9635177850723267, "reward_meter_std": 0.020120054483413696, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02012004144489765, "reward_total_composite_mean": 0.9635177850723267, "reward_total_composite_std": 0.020120054483413696, "reward_total_mean": 0.9635177850723267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9635177850723267, "rewards/meter/std": 0.020120054483413696, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9635177850723267, "rewards/total_composite/std": 0.020120054483413696, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0330264568328857, "sampling/importance_sampling_ratio/min": 0.5093486905097961, "sampling/sampling_logp_difference/max": 1.0730462074279785, "sampling/sampling_logp_difference/mean": 0.058272585272789, "step": 1484 }, { "clip_ratio/high_max": 0.025170997832901776, "clip_ratio/high_mean": 0.025170997832901776, "clip_ratio/low_mean": 0.006699939724057913, "clip_ratio/low_min": 0.006699939724057913, "clip_ratio/region_mean": 0.03187093755695969, "completions/clipped_ratio": 0.0, "completions/max_length": 95.0, "completions/max_terminated_length": 95.0, "completions/mean_length": 90.0, "completions/mean_terminated_length": 90.0, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.5103895664215088, "epoch": 0.05964574045065671, "frac_reward_zero_std": 0.0, "grad_norm": 4.295280456542969, "learning_rate": 5.5030303030303034e-06, "loss": 0.0024, "num_tokens": 3341484.0, "reward": 0.6436759233474731, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6956532597541809, "reward_meter_std": 0.21607494354248047, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.27830079197883606, "reward_total_composite_mean": 0.6436759233474731, "reward_total_composite_std": 0.27830082178115845, "reward_total_mean": 0.6436759233474731, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6956532597541809, "rewards/meter/std": 0.21607494354248047, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.6436759233474731, "rewards/total_composite/std": 0.27830082178115845, "sampling/importance_sampling_ratio/max": 1.6940116882324219, "sampling/importance_sampling_ratio/mean": 1.0068004131317139, "sampling/importance_sampling_ratio/min": 0.21599619090557098, "sampling/sampling_logp_difference/max": 1.5324945449829102, "sampling/sampling_logp_difference/mean": 0.07055231183767319, "step": 1485 }, { "clip_ratio/high_max": 0.010355054982937872, "clip_ratio/high_mean": 0.010355054982937872, "clip_ratio/low_mean": 0.021903468994423747, "clip_ratio/low_min": 0.021903468994423747, "clip_ratio/region_mean": 0.03225852397736162, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 85.875, "completions/mean_terminated_length": 85.875, "completions/min_length": 81.0, "completions/min_terminated_length": 81.0, "entropy": 0.6251698546111584, "epoch": 0.05968590593244166, "frac_reward_zero_std": 0.0, "grad_norm": 5.7835164070129395, "learning_rate": 5.500000000000001e-06, "loss": 0.0174, "num_tokens": 3343427.0, "reward": 0.41748714447021484, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5018972158432007, "reward_meter_std": 0.3999512195587158, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.3111114799976349, "reward_total_composite_mean": 0.41748714447021484, "reward_total_composite_std": 0.3111114799976349, "reward_total_mean": 0.41748714447021484, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5018972158432007, "rewards/meter/std": 0.3999512195587158, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.41748714447021484, "rewards/total_composite/std": 0.3111114799976349, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0213876962661743, "sampling/importance_sampling_ratio/min": 0.19992808997631073, "sampling/sampling_logp_difference/max": 1.609797477722168, "sampling/sampling_logp_difference/mean": 0.0595453642308712, "step": 1486 }, { "clip_ratio/high_max": 0.02728484314866364, "clip_ratio/high_mean": 0.02728484314866364, "clip_ratio/low_mean": 0.0132503192871809, "clip_ratio/low_min": 0.0132503192871809, "clip_ratio/region_mean": 0.04053516243584454, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 55.25, "completions/mean_terminated_length": 55.25, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.33666370064020157, "epoch": 0.05972607141422662, "frac_reward_zero_std": 0.0, "grad_norm": 18.545215606689453, "learning_rate": 5.496969696969697e-06, "loss": 0.0425, "num_tokens": 3345165.0, "reward": 0.9714435338973999, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9714435338973999, "reward_meter_std": 0.04950811713933945, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04950810596346855, "reward_total_composite_mean": 0.9714435338973999, "reward_total_composite_std": 0.04950811713933945, "reward_total_mean": 0.9714435338973999, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9714435338973999, "rewards/meter/std": 0.04950811713933945, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9714435338973999, "rewards/total_composite/std": 0.04950811713933945, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.014945149421692, "sampling/importance_sampling_ratio/min": 0.2995925843715668, "sampling/sampling_logp_difference/max": 1.205331802368164, "sampling/sampling_logp_difference/mean": 0.06842397898435593, "step": 1487 }, { "clip_ratio/high_max": 0.020023792050778866, "clip_ratio/high_mean": 0.020023792050778866, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/region_mean": 0.026968236546963453, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 55.625, "completions/mean_terminated_length": 55.625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.2505888659507036, "epoch": 0.05976623689601157, "frac_reward_zero_std": 0.0, "grad_norm": 4.115540504455566, "learning_rate": 5.493939393939395e-06, "loss": -0.0096, "num_tokens": 3346850.0, "reward": 0.992435097694397, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.992435097694397, "reward_meter_std": 0.006868826691061258, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006868841592222452, "reward_total_composite_mean": 0.992435097694397, "reward_total_composite_std": 0.006868826691061258, "reward_total_mean": 0.992435097694397, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.992435097694397, "rewards/meter/std": 0.006868826691061258, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992435097694397, "rewards/total_composite/std": 0.006868826691061258, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0044257640838623, "sampling/importance_sampling_ratio/min": 0.3714584410190582, "sampling/sampling_logp_difference/max": 1.0334584712982178, "sampling/sampling_logp_difference/mean": 0.0397869236767292, "step": 1488 }, { "clip_ratio/high_max": 0.04566098749637604, "clip_ratio/high_mean": 0.04566098749637604, "clip_ratio/low_mean": 0.013681166106835008, "clip_ratio/low_min": 0.013681166106835008, "clip_ratio/region_mean": 0.059342153603211045, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 90.875, "completions/mean_terminated_length": 90.875, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.6783212311565876, "epoch": 0.059806402377796525, "frac_reward_zero_std": 0.0, "grad_norm": 3.699056386947632, "learning_rate": 5.490909090909091e-06, "loss": 0.0041, "num_tokens": 3348897.0, "reward": 0.7304862141609192, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7553209066390991, "reward_meter_std": 0.27557846903800964, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.2595451772212982, "reward_total_composite_mean": 0.7304862141609192, "reward_total_composite_std": 0.2595451772212982, "reward_total_mean": 0.7304862141609192, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7553209066390991, "rewards/meter/std": 0.27557846903800964, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7304862141609192, "rewards/total_composite/std": 0.2595451772212982, "sampling/importance_sampling_ratio/max": 1.844269871711731, "sampling/importance_sampling_ratio/mean": 1.0141518115997314, "sampling/importance_sampling_ratio/min": 0.252288281917572, "sampling/sampling_logp_difference/max": 1.377182960510254, "sampling/sampling_logp_difference/mean": 0.07237864285707474, "step": 1489 }, { "clip_ratio/high_max": 0.03319170791655779, "clip_ratio/high_mean": 0.03319170791655779, "clip_ratio/low_mean": 0.03902717446908355, "clip_ratio/low_min": 0.03902717446908355, "clip_ratio/region_mean": 0.07221888238564134, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 155.625, "completions/mean_terminated_length": 155.625, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.8372621536254883, "epoch": 0.05984656785958148, "frac_reward_zero_std": 0.0, "grad_norm": 3.8729865550994873, "learning_rate": 5.487878787878789e-06, "loss": 0.019, "num_tokens": 3351734.0, "reward": 0.6489295363426208, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8863186240196228, "reward_meter_std": 0.23323532938957214, "reward_repeat_penalty_mean": 0.8888888955116272, "reward_repeat_penalty_std": 0.10286889225244522, "reward_std": 0.17390431463718414, "reward_total_composite_mean": 0.6489295363426208, "reward_total_composite_std": 0.17390431463718414, "reward_total_mean": 0.6489295363426208, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8863186240196228, "rewards/meter/std": 0.23323532938957214, "rewards/repeat_penalty/mean": 0.8888888955116272, "rewards/repeat_penalty/std": 0.10286889225244522, "rewards/total_composite/mean": 0.6489295363426208, "rewards/total_composite/std": 0.17390431463718414, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.026315450668335, "sampling/importance_sampling_ratio/min": 0.24719087779521942, "sampling/sampling_logp_difference/max": 1.3975944519042969, "sampling/sampling_logp_difference/mean": 0.08673720061779022, "step": 1490 }, { "clip_ratio/high_max": 0.02461273316293955, "clip_ratio/high_mean": 0.02461273316293955, "clip_ratio/low_mean": 0.013612689450383186, "clip_ratio/low_min": 0.013612689450383186, "clip_ratio/region_mean": 0.038225422613322735, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.31265123561024666, "epoch": 0.05988673334136643, "frac_reward_zero_std": 0.0, "grad_norm": 5.608306884765625, "learning_rate": 5.484848484848485e-06, "loss": -0.0074, "num_tokens": 3353757.0, "reward": 0.9181324243545532, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9181324243545532, "reward_meter_std": 0.15554846823215485, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.15554848313331604, "reward_total_composite_mean": 0.9181324243545532, "reward_total_composite_std": 0.15554846823215485, "reward_total_mean": 0.9181324243545532, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9181324243545532, "rewards/meter/std": 0.15554846823215485, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9181324243545532, "rewards/total_composite/std": 0.15554846823215485, "sampling/importance_sampling_ratio/max": 1.6148149967193604, "sampling/importance_sampling_ratio/mean": 1.010719895362854, "sampling/importance_sampling_ratio/min": 0.2553488314151764, "sampling/sampling_logp_difference/max": 1.3651247024536133, "sampling/sampling_logp_difference/mean": 0.04674442857503891, "step": 1491 }, { "clip_ratio/high_max": 0.019893484190106392, "clip_ratio/high_mean": 0.019893484190106392, "clip_ratio/low_mean": 0.025138093391433358, "clip_ratio/low_min": 0.025138093391433358, "clip_ratio/region_mean": 0.04503157758153975, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 56.625, "completions/mean_terminated_length": 56.625, "completions/min_length": 53.0, "completions/min_terminated_length": 53.0, "entropy": 0.9116378463804722, "epoch": 0.05992689882315139, "frac_reward_zero_std": 0.0, "grad_norm": 7.214973449707031, "learning_rate": 5.4818181818181825e-06, "loss": -0.0022, "num_tokens": 3355426.0, "reward": 0.7181947231292725, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7181947231292725, "reward_meter_std": 0.3745432198047638, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3745432198047638, "reward_total_composite_mean": 0.7181947231292725, "reward_total_composite_std": 0.3745432198047638, "reward_total_mean": 0.7181947231292725, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7181947231292725, "rewards/meter/std": 0.3745432198047638, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7181947231292725, "rewards/total_composite/std": 0.3745432198047638, "sampling/importance_sampling_ratio/max": 1.654712200164795, "sampling/importance_sampling_ratio/mean": 1.0328161716461182, "sampling/importance_sampling_ratio/min": 0.2746668756008148, "sampling/sampling_logp_difference/max": 1.292196273803711, "sampling/sampling_logp_difference/mean": 0.07947485893964767, "step": 1492 }, { "clip_ratio/high_max": 0.03274800395593047, "clip_ratio/high_mean": 0.03274800395593047, "clip_ratio/low_mean": 0.011852287454530597, "clip_ratio/low_min": 0.011852287454530597, "clip_ratio/region_mean": 0.04460029141046107, "completions/clipped_ratio": 0.0, "completions/max_length": 259.0, "completions/max_terminated_length": 259.0, "completions/mean_length": 235.875, "completions/mean_terminated_length": 235.875, "completions/min_length": 211.0, "completions/min_terminated_length": 211.0, "entropy": 0.8409713134169579, "epoch": 0.05996706430493634, "frac_reward_zero_std": 0.0, "grad_norm": 4.562529563903809, "learning_rate": 5.478787878787879e-06, "loss": 0.0529, "num_tokens": 3358873.0, "reward": 0.5754737257957458, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.7638888955116272, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.9776490926742554, "reward_meter_std": 0.024429455399513245, "reward_repeat_penalty_mean": 0.8934294581413269, "reward_repeat_penalty_std": 0.1227281242609024, "reward_std": 0.2489534169435501, "reward_total_composite_mean": 0.5754737257957458, "reward_total_composite_std": 0.2489534169435501, "reward_total_mean": 0.5754737257957458, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.7638888955116272, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.9776490926742554, "rewards/meter/std": 0.024429455399513245, "rewards/repeat_penalty/mean": 0.8934294581413269, "rewards/repeat_penalty/std": 0.1227281242609024, "rewards/total_composite/mean": 0.5754737257957458, "rewards/total_composite/std": 0.2489534169435501, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0195549726486206, "sampling/importance_sampling_ratio/min": 0.13726016879081726, "sampling/sampling_logp_difference/max": 1.9858770370483398, "sampling/sampling_logp_difference/mean": 0.07718866318464279, "step": 1493 }, { "clip_ratio/high_max": 0.014402368804439902, "clip_ratio/high_mean": 0.014402368804439902, "clip_ratio/low_mean": 0.028380894218571484, "clip_ratio/low_min": 0.028380894218571484, "clip_ratio/region_mean": 0.042783263023011386, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 85.375, "completions/mean_terminated_length": 85.375, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 1.0446771308779716, "epoch": 0.060007229786721294, "frac_reward_zero_std": 0.0, "grad_norm": 7.282406806945801, "learning_rate": 5.475757575757576e-06, "loss": 0.0076, "num_tokens": 3360884.0, "reward": 0.6164090037345886, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.665711522102356, "reward_meter_std": 0.4568917751312256, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.4254213869571686, "reward_total_composite_mean": 0.6164090037345886, "reward_total_composite_std": 0.42542141675949097, "reward_total_mean": 0.6164090037345886, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.665711522102356, "rewards/meter/std": 0.4568917751312256, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6164090037345886, "rewards/total_composite/std": 0.42542141675949097, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.025902509689331, "sampling/importance_sampling_ratio/min": 0.19224660098552704, "sampling/sampling_logp_difference/max": 1.6489763259887695, "sampling/sampling_logp_difference/mean": 0.09246491640806198, "step": 1494 }, { "clip_ratio/high_max": 0.011480186600238085, "clip_ratio/high_mean": 0.011480186600238085, "clip_ratio/low_mean": 0.007462686393409967, "clip_ratio/low_min": 0.007462686393409967, "clip_ratio/region_mean": 0.018942872993648052, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.625, "completions/mean_terminated_length": 65.625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.10151981003582478, "epoch": 0.06004739526850625, "frac_reward_zero_std": 0.0, "grad_norm": 3.7021563053131104, "learning_rate": 5.472727272727273e-06, "loss": 0.0083, "num_tokens": 3362761.0, "reward": 0.9978386163711548, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978386163711548, "reward_meter_std": 0.0011219180887565017, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0011219230946153402, "reward_total_composite_mean": 0.9978386163711548, "reward_total_composite_std": 0.0011219180887565017, "reward_total_mean": 0.9978386163711548, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978386163711548, "rewards/meter/std": 0.0011219180887565017, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978386163711548, "rewards/total_composite/std": 0.0011219180887565017, "sampling/importance_sampling_ratio/max": 1.4048590660095215, "sampling/importance_sampling_ratio/mean": 1.000672698020935, "sampling/importance_sampling_ratio/min": 0.5120251774787903, "sampling/sampling_logp_difference/max": 0.6693814992904663, "sampling/sampling_logp_difference/mean": 0.019154906272888184, "step": 1495 }, { "clip_ratio/high_max": 0.02847922092769295, "clip_ratio/high_mean": 0.02847922092769295, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/region_mean": 0.030402297852560878, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 65.75, "completions/mean_terminated_length": 65.75, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.3818973954766989, "epoch": 0.0600875607502912, "frac_reward_zero_std": 0.0, "grad_norm": 4.584413528442383, "learning_rate": 5.469696969696971e-06, "loss": 0.0026, "num_tokens": 3364575.0, "reward": 0.9560348987579346, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9560348987579346, "reward_meter_std": 0.09728128463029861, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09728129208087921, "reward_total_composite_mean": 0.9560348987579346, "reward_total_composite_std": 0.09728128463029861, "reward_total_mean": 0.9560348987579346, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9560348987579346, "rewards/meter/std": 0.09728128463029861, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9560348987579346, "rewards/total_composite/std": 0.09728128463029861, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0161854028701782, "sampling/importance_sampling_ratio/min": 0.3192536234855652, "sampling/sampling_logp_difference/max": 1.1417694091796875, "sampling/sampling_logp_difference/mean": 0.04986903816461563, "step": 1496 }, { "clip_ratio/high_max": 0.05977132357656956, "clip_ratio/high_mean": 0.05977132357656956, "clip_ratio/low_mean": 0.01920653972774744, "clip_ratio/low_min": 0.01920653972774744, "clip_ratio/region_mean": 0.078977863304317, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 89.125, "completions/mean_terminated_length": 89.125, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.7317456901073456, "epoch": 0.060127726232076156, "frac_reward_zero_std": 0.0, "grad_norm": 6.656605243682861, "learning_rate": 5.466666666666667e-06, "loss": 0.0391, "num_tokens": 3366568.0, "reward": 0.5929021239280701, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6094886064529419, "reward_meter_std": 0.4335397183895111, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.4219600260257721, "reward_total_composite_mean": 0.5929021239280701, "reward_total_composite_std": 0.4219600558280945, "reward_total_mean": 0.5929021239280701, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6094886064529419, "rewards/meter/std": 0.4335397183895111, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.5929021239280701, "rewards/total_composite/std": 0.4219600558280945, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0115280151367188, "sampling/importance_sampling_ratio/min": 0.18578758835792542, "sampling/sampling_logp_difference/max": 1.6831512451171875, "sampling/sampling_logp_difference/mean": 0.08898783475160599, "step": 1497 }, { "clip_ratio/high_max": 0.017676767893135548, "clip_ratio/high_mean": 0.017676767893135548, "clip_ratio/low_mean": 0.03597753681242466, "clip_ratio/low_min": 0.03597753681242466, "clip_ratio/region_mean": 0.05365430470556021, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 34.625, "completions/mean_terminated_length": 34.625, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.3656330443918705, "epoch": 0.06016789171386111, "frac_reward_zero_std": 0.0, "grad_norm": 9.448737144470215, "learning_rate": 5.463636363636364e-06, "loss": 0.0331, "num_tokens": 3368045.0, "reward": 0.9881667494773865, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9881667494773865, "reward_meter_std": 0.003932104911655188, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003932103049010038, "reward_total_composite_mean": 0.9881667494773865, "reward_total_composite_std": 0.003932104911655188, "reward_total_mean": 0.9881667494773865, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9881667494773865, "rewards/meter/std": 0.003932104911655188, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9881667494773865, "rewards/total_composite/std": 0.003932104911655188, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.021955132484436, "sampling/importance_sampling_ratio/min": 0.19301611185073853, "sampling/sampling_logp_difference/max": 1.6449816226959229, "sampling/sampling_logp_difference/mean": 0.04901129752397537, "step": 1498 }, { "clip_ratio/high_max": 0.05129538010805845, "clip_ratio/high_mean": 0.05129538010805845, "clip_ratio/low_mean": 0.024429237004369497, "clip_ratio/low_min": 0.024429237004369497, "clip_ratio/region_mean": 0.07572461711242795, "completions/clipped_ratio": 0.0, "completions/max_length": 135.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 128.5, "completions/mean_terminated_length": 128.5, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.9432215169072151, "epoch": 0.060208057195646064, "frac_reward_zero_std": 0.0, "grad_norm": 3.978285551071167, "learning_rate": 5.460606060606061e-06, "loss": 0.0209, "num_tokens": 3370801.0, "reward": 0.9302494525909424, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9624004364013672, "reward_meter_std": 0.06776177138090134, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.10968230664730072, "reward_total_composite_mean": 0.9302494525909424, "reward_total_composite_std": 0.10968229919672012, "reward_total_mean": 0.9302494525909424, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9624004364013672, "rewards/meter/std": 0.06776177138090134, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9302494525909424, "rewards/total_composite/std": 0.10968229919672012, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.029836893081665, "sampling/importance_sampling_ratio/min": 0.27643805742263794, "sampling/sampling_logp_difference/max": 1.2857685089111328, "sampling/sampling_logp_difference/mean": 0.09097252786159515, "step": 1499 }, { "clip_ratio/high_max": 0.01567198208067566, "clip_ratio/high_mean": 0.01567198208067566, "clip_ratio/low_mean": 0.02421415294520557, "clip_ratio/low_min": 0.02421415294520557, "clip_ratio/region_mean": 0.03988613502588123, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 63.75, "completions/mean_terminated_length": 63.75, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.7506604716181755, "epoch": 0.06024822267743102, "frac_reward_zero_std": 0.0, "grad_norm": 5.225851058959961, "learning_rate": 5.457575757575758e-06, "loss": -0.0122, "num_tokens": 3372511.0, "reward": 0.9959254264831543, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959254264831543, "reward_meter_std": 0.0024275805335491896, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002427596366032958, "reward_total_composite_mean": 0.9959254264831543, "reward_total_composite_std": 0.0024275805335491896, "reward_total_mean": 0.9959254264831543, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959254264831543, "rewards/meter/std": 0.0024275805335491896, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9959254264831543, "rewards/total_composite/std": 0.0024275805335491896, "sampling/importance_sampling_ratio/max": 1.7771986722946167, "sampling/importance_sampling_ratio/mean": 1.0106977224349976, "sampling/importance_sampling_ratio/min": 0.40423041582107544, "sampling/sampling_logp_difference/max": 0.9057703018188477, "sampling/sampling_logp_difference/mean": 0.07604724913835526, "step": 1500 }, { "epoch": 0.06024822267743102, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/max_length": 304.38461538461536, "eval_completions/max_terminated_length": 271.9230769230769, "eval_completions/mean_length": 163.32692307692307, "eval_completions/mean_terminated_length": 156.44917766864484, "eval_completions/min_length": 57.92307692307692, "eval_completions/min_terminated_length": 57.92307692307692, "eval_entropy": 0.5243608149198385, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 3372511.0, "eval_reward": 0.4787046152811784, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_count_adherence_mean": 0.8234805831542382, "eval_reward_count_adherence_std": 0.17093561990902975, "eval_reward_meter_mean": 0.6285286339429709, "eval_reward_meter_std": 0.44250402083763707, "eval_reward_repeat_penalty_mean": 0.8920627649013813, "eval_reward_repeat_penalty_std": 0.12081348953338769, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.4787046152811784, "eval_reward_total_composite_std": 0.3746733757165762, "eval_reward_total_mean": 0.4787046152811784, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/count_adherence/mean": 0.8234805831542382, "eval_rewards/count_adherence/std": 0.17093561990902975, "eval_rewards/meter/mean": 0.6285286339429709, "eval_rewards/meter/std": 0.44250402083763707, "eval_rewards/repeat_penalty/mean": 0.8920627649013813, "eval_rewards/repeat_penalty/std": 0.12081348953338769, "eval_rewards/total_composite/mean": 0.4787046152811784, "eval_rewards/total_composite/std": 0.3746733757165762, "eval_runtime": 59.2586, "eval_samples_per_second": 1.755, "eval_sampling/importance_sampling_ratio/max": 1.5625533782518828, "eval_sampling/importance_sampling_ratio/mean": 1.0148254266152015, "eval_sampling/importance_sampling_ratio/min": 0.32900666961303127, "eval_sampling/sampling_logp_difference/max": 1.12705293068519, "eval_sampling/sampling_logp_difference/mean": 0.04665847968023557, "eval_steps_per_second": 0.219, "step": 1500 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 33.0, "completions/mean_terminated_length": 33.0, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.031635227147489786, "epoch": 0.06028838815921597, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.4545454545454545e-06, "loss": 0.0, "num_tokens": 3374031.0, "reward": 0.9988001585006714, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988001585006714, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9988001585006714, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9988001585006714, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988001585006714, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988001585006714, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0545777082443237, "sampling/importance_sampling_ratio/mean": 1.0031137466430664, "sampling/importance_sampling_ratio/min": 0.9900038242340088, "sampling/sampling_logp_difference/max": 0.05314040184020996, "sampling/sampling_logp_difference/mean": 0.0034284384455531836, "step": 1501 }, { "clip_ratio/high_max": 0.028100816532969475, "clip_ratio/high_mean": 0.028100816532969475, "clip_ratio/low_mean": 0.020408162847161293, "clip_ratio/low_min": 0.020408162847161293, "clip_ratio/region_mean": 0.04850897938013077, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 98.125, "completions/mean_terminated_length": 98.125, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.5762354806065559, "epoch": 0.060328553641000926, "frac_reward_zero_std": 0.0, "grad_norm": 5.628778457641602, "learning_rate": 5.451515151515152e-06, "loss": 0.0069, "num_tokens": 3376224.0, "reward": 0.8838797807693481, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8838797807693481, "reward_meter_std": 0.2619064450263977, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2619064748287201, "reward_total_composite_mean": 0.8838797807693481, "reward_total_composite_std": 0.2619064450263977, "reward_total_mean": 0.8838797807693481, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8838797807693481, "rewards/meter/std": 0.2619064450263977, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8838797807693481, "rewards/total_composite/std": 0.2619064450263977, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0124897956848145, "sampling/importance_sampling_ratio/min": 0.21037398278713226, "sampling/sampling_logp_difference/max": 1.558868408203125, "sampling/sampling_logp_difference/mean": 0.06008461117744446, "step": 1502 }, { "clip_ratio/high_max": 0.001923076924867928, "clip_ratio/high_mean": 0.001923076924867928, "clip_ratio/low_mean": 0.009538664249703288, "clip_ratio/low_min": 0.009538664249703288, "clip_ratio/region_mean": 0.011461741174571216, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 65.5, "completions/mean_terminated_length": 65.5, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.10324481781572104, "epoch": 0.06036871912278588, "frac_reward_zero_std": 0.0, "grad_norm": 2.2546792030334473, "learning_rate": 5.448484848484848e-06, "loss": -0.0001, "num_tokens": 3378172.0, "reward": 0.998626708984375, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998626708984375, "reward_meter_std": 0.00024546196800656617, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002454565546941012, "reward_total_composite_mean": 0.998626708984375, "reward_total_composite_std": 0.00024546196800656617, "reward_total_mean": 0.998626708984375, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998626708984375, "rewards/meter/std": 0.00024546196800656617, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998626708984375, "rewards/total_composite/std": 0.00024546196800656617, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001434326171875, "sampling/importance_sampling_ratio/min": 0.2059563547372818, "sampling/sampling_logp_difference/max": 1.5800909996032715, "sampling/sampling_logp_difference/mean": 0.021490152925252914, "step": 1503 }, { "clip_ratio/high_max": 0.019740989664569497, "clip_ratio/high_mean": 0.019740989664569497, "clip_ratio/low_mean": 0.008970056660473347, "clip_ratio/low_min": 0.008970056660473347, "clip_ratio/region_mean": 0.028711046325042844, "completions/clipped_ratio": 0.0, "completions/max_length": 408.0, "completions/max_terminated_length": 408.0, "completions/mean_length": 384.625, "completions/mean_terminated_length": 384.625, "completions/min_length": 366.0, "completions/min_terminated_length": 366.0, "entropy": 0.44889700040221214, "epoch": 0.060408884604570834, "frac_reward_zero_std": 0.0, "grad_norm": 2.0835001468658447, "learning_rate": 5.445454545454546e-06, "loss": -0.0139, "num_tokens": 3383481.0, "reward": 0.4811665415763855, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6041666269302368, "reward_count_adherence_std": 0.01964184269309044, "reward_meter_mean": 0.9722083806991577, "reward_meter_std": 0.05760227516293526, "reward_repeat_penalty_mean": 0.8138812780380249, "reward_repeat_penalty_std": 0.13613860309123993, "reward_std": 0.10122059285640717, "reward_total_composite_mean": 0.4811665415763855, "reward_total_composite_std": 0.10122059285640717, "reward_total_mean": 0.4811665415763855, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6041666269302368, "rewards/count_adherence/std": 0.01964184269309044, "rewards/meter/mean": 0.9722083806991577, "rewards/meter/std": 0.05760227516293526, "rewards/repeat_penalty/mean": 0.8138812780380249, "rewards/repeat_penalty/std": 0.13613860309123993, "rewards/total_composite/mean": 0.4811665415763855, "rewards/total_composite/std": 0.10122059285640717, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0164600610733032, "sampling/importance_sampling_ratio/min": 0.19235540926456451, "sampling/sampling_logp_difference/max": 1.6484105587005615, "sampling/sampling_logp_difference/mean": 0.04671372473239899, "step": 1504 }, { "clip_ratio/high_max": 0.035130204167217016, "clip_ratio/high_mean": 0.035130204167217016, "clip_ratio/low_mean": 0.006312201963737607, "clip_ratio/low_min": 0.006312201963737607, "clip_ratio/region_mean": 0.04144240613095462, "completions/clipped_ratio": 0.0, "completions/max_length": 267.0, "completions/max_terminated_length": 267.0, "completions/mean_length": 249.875, "completions/mean_terminated_length": 249.875, "completions/min_length": 232.0, "completions/min_terminated_length": 232.0, "entropy": 0.4347769767045975, "epoch": 0.06044905008635579, "frac_reward_zero_std": 0.0, "grad_norm": 2.243826389312744, "learning_rate": 5.442424242424243e-06, "loss": -0.0283, "num_tokens": 3387080.0, "reward": 0.584118127822876, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0534522607922554, "reward_meter_mean": 0.9895928502082825, "reward_meter_std": 0.008481908589601517, "reward_repeat_penalty_mean": 0.8847527503967285, "reward_repeat_penalty_std": 0.05292898043990135, "reward_std": 0.24487483501434326, "reward_total_composite_mean": 0.584118127822876, "reward_total_composite_std": 0.24487483501434326, "reward_total_mean": 0.584118127822876, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0534522607922554, "rewards/meter/mean": 0.9895928502082825, "rewards/meter/std": 0.008481908589601517, "rewards/repeat_penalty/mean": 0.8847527503967285, "rewards/repeat_penalty/std": 0.05292898043990135, "rewards/total_composite/mean": 0.584118127822876, "rewards/total_composite/std": 0.24487483501434326, "sampling/importance_sampling_ratio/max": 1.8827742338180542, "sampling/importance_sampling_ratio/mean": 1.0099492073059082, "sampling/importance_sampling_ratio/min": 0.2247946411371231, "sampling/sampling_logp_difference/max": 1.492568016052246, "sampling/sampling_logp_difference/mean": 0.045951321721076965, "step": 1505 }, { "clip_ratio/high_max": 0.005710955825634301, "clip_ratio/high_mean": 0.005710955825634301, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.007604895276017487, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.06401140103116632, "epoch": 0.06048921556814074, "frac_reward_zero_std": 0.0, "grad_norm": 2.0182154178619385, "learning_rate": 5.43939393939394e-06, "loss": 0.0015, "num_tokens": 3388903.0, "reward": 0.9988133907318115, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988133907318115, "reward_meter_std": 0.00020309248066041619, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020310519903432578, "reward_total_composite_mean": 0.9988133907318115, "reward_total_composite_std": 0.00020309248066041619, "reward_total_mean": 0.9988133907318115, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988133907318115, "rewards/meter/std": 0.00020309248066041619, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988133907318115, "rewards/total_composite/std": 0.00020309248066041619, "sampling/importance_sampling_ratio/max": 1.8056702613830566, "sampling/importance_sampling_ratio/mean": 1.0014187097549438, "sampling/importance_sampling_ratio/min": 0.5476920008659363, "sampling/sampling_logp_difference/max": 0.6020421981811523, "sampling/sampling_logp_difference/mean": 0.009212362580001354, "step": 1506 }, { "clip_ratio/high_max": 0.04194163717329502, "clip_ratio/high_mean": 0.04194163717329502, "clip_ratio/low_mean": 0.006993599818088114, "clip_ratio/low_min": 0.006993599818088114, "clip_ratio/region_mean": 0.048935236991383135, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 125.5, "completions/mean_terminated_length": 125.5, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.6917183268815279, "epoch": 0.060529381049925696, "frac_reward_zero_std": 0.0, "grad_norm": 3.6379973888397217, "learning_rate": 5.436363636363636e-06, "loss": 0.0083, "num_tokens": 3391499.0, "reward": 0.9441857933998108, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976436495780945, "reward_meter_std": 0.0009433169034309685, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07358825951814651, "reward_total_composite_mean": 0.9441857933998108, "reward_total_composite_std": 0.07358824461698532, "reward_total_mean": 0.9441857933998108, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976436495780945, "rewards/meter/std": 0.0009433169034309685, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9441857933998108, "rewards/total_composite/std": 0.07358824461698532, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0165196657180786, "sampling/importance_sampling_ratio/min": 0.2394247055053711, "sampling/sampling_logp_difference/max": 1.429516315460205, "sampling/sampling_logp_difference/mean": 0.07620169967412949, "step": 1507 }, { "clip_ratio/high_max": 0.050818526186048985, "clip_ratio/high_mean": 0.050818526186048985, "clip_ratio/low_mean": 0.018996416125446558, "clip_ratio/low_min": 0.018996416125446558, "clip_ratio/region_mean": 0.06981494231149554, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 90.625, "completions/mean_terminated_length": 90.625, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 1.3208992555737495, "epoch": 0.06056954653171065, "frac_reward_zero_std": 0.0, "grad_norm": 5.184098243713379, "learning_rate": 5.4333333333333335e-06, "loss": 0.0189, "num_tokens": 3393400.0, "reward": 0.9280390739440918, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9280390739440918, "reward_meter_std": 0.12549975514411926, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.12549975514411926, "reward_total_composite_mean": 0.9280390739440918, "reward_total_composite_std": 0.12549975514411926, "reward_total_mean": 0.9280390739440918, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9280390739440918, "rewards/meter/std": 0.12549975514411926, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9280390739440918, "rewards/total_composite/std": 0.12549975514411926, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.030521035194397, "sampling/importance_sampling_ratio/min": 0.2036202847957611, "sampling/sampling_logp_difference/max": 1.5914983749389648, "sampling/sampling_logp_difference/mean": 0.11991255730390549, "step": 1508 }, { "clip_ratio/high_max": 0.015109836007468402, "clip_ratio/high_mean": 0.015109836007468402, "clip_ratio/low_mean": 0.008413461968302727, "clip_ratio/low_min": 0.008413461968302727, "clip_ratio/region_mean": 0.02352329797577113, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 113.5, "completions/mean_terminated_length": 113.5, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.22751092724502087, "epoch": 0.060609712013495604, "frac_reward_zero_std": 0.0, "grad_norm": 5.793516159057617, "learning_rate": 5.430303030303032e-06, "loss": -0.0186, "num_tokens": 3395804.0, "reward": 0.8047507405281067, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9283153414726257, "reward_meter_std": 0.17925596237182617, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.1079898476600647, "reward_std": 0.20624437928199768, "reward_total_composite_mean": 0.8047507405281067, "reward_total_composite_std": 0.20624437928199768, "reward_total_mean": 0.8047507405281067, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9283153414726257, "rewards/meter/std": 0.17925596237182617, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.1079898476600647, "rewards/total_composite/mean": 0.8047507405281067, "rewards/total_composite/std": 0.20624437928199768, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0089442729949951, "sampling/importance_sampling_ratio/min": 0.29910770058631897, "sampling/sampling_logp_difference/max": 1.20695161819458, "sampling/sampling_logp_difference/mean": 0.03325022757053375, "step": 1509 }, { "clip_ratio/high_max": 0.023895947029814124, "clip_ratio/high_mean": 0.023895947029814124, "clip_ratio/low_mean": 0.010469448054209352, "clip_ratio/low_min": 0.010469448054209352, "clip_ratio/region_mean": 0.034365395084023476, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 57.875, "completions/mean_terminated_length": 57.875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.4132865481078625, "epoch": 0.06064987749528056, "frac_reward_zero_std": 0.0, "grad_norm": 6.951155662536621, "learning_rate": 5.427272727272728e-06, "loss": 0.0003, "num_tokens": 3397499.0, "reward": 0.7507508397102356, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7507508397102356, "reward_meter_std": 0.3587440252304077, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3587440550327301, "reward_total_composite_mean": 0.7507508397102356, "reward_total_composite_std": 0.3587440252304077, "reward_total_mean": 0.7507508397102356, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7507508397102356, "rewards/meter/std": 0.3587440252304077, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7507508397102356, "rewards/total_composite/std": 0.3587440252304077, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.015701174736023, "sampling/importance_sampling_ratio/min": 0.20673415064811707, "sampling/sampling_logp_difference/max": 1.5763216018676758, "sampling/sampling_logp_difference/mean": 0.053551964461803436, "step": 1510 }, { "clip_ratio/high_max": 0.022818502504378557, "clip_ratio/high_mean": 0.022818502504378557, "clip_ratio/low_mean": 0.02230392163619399, "clip_ratio/low_min": 0.02230392163619399, "clip_ratio/region_mean": 0.04512242414057255, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 107.125, "completions/mean_terminated_length": 49.28571701049805, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.5180366039276123, "epoch": 0.06069004297706551, "frac_reward_zero_std": 0.0, "grad_norm": 3.092373847961426, "learning_rate": 5.424242424242425e-06, "loss": -0.0464, "num_tokens": 3399092.0, "reward": 0.42717820405960083, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.42876946926116943, "reward_meter_std": 0.4184267520904541, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.420255184173584, "reward_total_composite_mean": 0.42717820405960083, "reward_total_composite_std": 0.4202551543712616, "reward_total_mean": 0.42717820405960083, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.42876946926116943, "rewards/meter/std": 0.4184267520904541, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.42717820405960083, "rewards/total_composite/std": 0.4202551543712616, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0123780965805054, "sampling/importance_sampling_ratio/min": 0.24309387803077698, "sampling/sampling_logp_difference/max": 1.4143075942993164, "sampling/sampling_logp_difference/mean": 0.07508043199777603, "step": 1511 }, { "clip_ratio/high_max": 0.031007058219984174, "clip_ratio/high_mean": 0.031007058219984174, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.031007058219984174, "completions/clipped_ratio": 0.375, "completions/max_length": 512.0, "completions/max_terminated_length": 367.0, "completions/mean_length": 405.75, "completions/mean_terminated_length": 342.0, "completions/min_length": 309.0, "completions/min_terminated_length": 309.0, "entropy": 0.5392166823148727, "epoch": 0.060730208458850465, "frac_reward_zero_std": 0.0, "grad_norm": 1.4069581031799316, "learning_rate": 5.421212121212122e-06, "loss": -0.3845, "num_tokens": 3402818.0, "reward": 0.33205440640449524, "reward_arabic_clean_mean": 0.625, "reward_arabic_clean_std": 0.5175492167472839, "reward_count_adherence_mean": 0.6102941036224365, "reward_count_adherence_std": 0.08858475089073181, "reward_meter_mean": 0.7066450119018555, "reward_meter_std": 0.3806198835372925, "reward_repeat_penalty_mean": 0.9273183345794678, "reward_repeat_penalty_std": 0.11418548226356506, "reward_std": 0.27961447834968567, "reward_total_composite_mean": 0.33205440640449524, "reward_total_composite_std": 0.27961450815200806, "reward_total_mean": 0.33205440640449524, "rewards/arabic_clean/mean": 0.625, "rewards/arabic_clean/std": 0.5175492167472839, "rewards/count_adherence/mean": 0.6102941036224365, "rewards/count_adherence/std": 0.08858475089073181, "rewards/meter/mean": 0.7066450119018555, "rewards/meter/std": 0.3806198835372925, "rewards/repeat_penalty/mean": 0.9273183345794678, "rewards/repeat_penalty/std": 0.11418548226356506, "rewards/total_composite/mean": 0.33205440640449524, "rewards/total_composite/std": 0.27961450815200806, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.020935297012329, "sampling/importance_sampling_ratio/min": 0.1479973942041397, "sampling/sampling_logp_difference/max": 1.9105606079101562, "sampling/sampling_logp_difference/mean": 0.08264101296663284, "step": 1512 }, { "clip_ratio/high_max": 0.014931121026165783, "clip_ratio/high_mean": 0.014931121026165783, "clip_ratio/low_mean": 0.005488860071636736, "clip_ratio/low_min": 0.005488860071636736, "clip_ratio/region_mean": 0.02041998109780252, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2136769648641348, "epoch": 0.06077037394063542, "frac_reward_zero_std": 0.0, "grad_norm": 4.46673583984375, "learning_rate": 5.418181818181819e-06, "loss": 0.0171, "num_tokens": 3404619.0, "reward": 0.9849165081977844, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9849165081977844, "reward_meter_std": 0.019791634753346443, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.019791632890701294, "reward_total_composite_mean": 0.9849165081977844, "reward_total_composite_std": 0.019791634753346443, "reward_total_mean": 0.9849165081977844, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9849165081977844, "rewards/meter/std": 0.019791634753346443, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9849165081977844, "rewards/total_composite/std": 0.019791634753346443, "sampling/importance_sampling_ratio/max": 1.5401774644851685, "sampling/importance_sampling_ratio/mean": 1.0061801671981812, "sampling/importance_sampling_ratio/min": 0.5287384390830994, "sampling/sampling_logp_difference/max": 0.6372613906860352, "sampling/sampling_logp_difference/mean": 0.025447804480791092, "step": 1513 }, { "clip_ratio/high_max": 0.04533929843455553, "clip_ratio/high_mean": 0.04533929843455553, "clip_ratio/low_mean": 0.012164860963821411, "clip_ratio/low_min": 0.012164860963821411, "clip_ratio/region_mean": 0.05750415939837694, "completions/clipped_ratio": 0.0, "completions/max_length": 236.0, "completions/max_terminated_length": 236.0, "completions/mean_length": 228.0, "completions/mean_terminated_length": 228.0, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 1.0589845478534698, "epoch": 0.06081053942242037, "frac_reward_zero_std": 0.0, "grad_norm": 2.8653032779693604, "learning_rate": 5.415151515151515e-06, "loss": -0.0047, "num_tokens": 3408083.0, "reward": 0.7759657502174377, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9303399324417114, "reward_meter_std": 0.1539093255996704, "reward_repeat_penalty_mean": 0.9519230723381042, "reward_repeat_penalty_std": 0.05723259598016739, "reward_std": 0.14068925380706787, "reward_total_composite_mean": 0.7759657502174377, "reward_total_composite_std": 0.14068926870822906, "reward_total_mean": 0.7759657502174377, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9303399324417114, "rewards/meter/std": 0.1539093255996704, "rewards/repeat_penalty/mean": 0.9519230723381042, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.7759657502174377, "rewards/total_composite/std": 0.14068926870822906, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0278751850128174, "sampling/importance_sampling_ratio/min": 0.16584692895412445, "sampling/sampling_logp_difference/max": 1.7966899871826172, "sampling/sampling_logp_difference/mean": 0.08908653259277344, "step": 1514 }, { "clip_ratio/high_max": 0.0221144916722551, "clip_ratio/high_mean": 0.0221144916722551, "clip_ratio/low_mean": 0.0055147059028968215, "clip_ratio/low_min": 0.0055147059028968215, "clip_ratio/region_mean": 0.02762919757515192, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2124041486531496, "epoch": 0.06085070490420533, "frac_reward_zero_std": 0.0, "grad_norm": 3.14272403717041, "learning_rate": 5.412121212121213e-06, "loss": -0.0027, "num_tokens": 3410076.0, "reward": 0.9943439960479736, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943439960479736, "reward_meter_std": 0.0008108045440167189, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008108008187264204, "reward_total_composite_mean": 0.9943439960479736, "reward_total_composite_std": 0.0008108045440167189, "reward_total_mean": 0.9943439960479736, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943439960479736, "rewards/meter/std": 0.0008108045440167189, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943439960479736, "rewards/total_composite/std": 0.0008108045440167189, "sampling/importance_sampling_ratio/max": 1.5258408784866333, "sampling/importance_sampling_ratio/mean": 1.0022821426391602, "sampling/importance_sampling_ratio/min": 0.39603835344314575, "sampling/sampling_logp_difference/max": 0.9262442588806152, "sampling/sampling_logp_difference/mean": 0.03039509244263172, "step": 1515 }, { "clip_ratio/high_max": 0.009246049099601805, "clip_ratio/high_mean": 0.009246049099601805, "clip_ratio/low_mean": 0.003818796598352492, "clip_ratio/low_min": 0.003818796598352492, "clip_ratio/region_mean": 0.013064845697954297, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.19451193884015083, "epoch": 0.06089087038599028, "frac_reward_zero_std": 0.0, "grad_norm": 6.165884017944336, "learning_rate": 5.409090909090909e-06, "loss": -0.0086, "num_tokens": 3412012.0, "reward": 0.9924481511116028, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9924481511116028, "reward_meter_std": 0.005331422667950392, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0053314100950956345, "reward_total_composite_mean": 0.9924481511116028, "reward_total_composite_std": 0.005331422667950392, "reward_total_mean": 0.9924481511116028, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9924481511116028, "rewards/meter/std": 0.005331422667950392, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924481511116028, "rewards/total_composite/std": 0.005331422667950392, "sampling/importance_sampling_ratio/max": 1.659652829170227, "sampling/importance_sampling_ratio/mean": 1.0072860717773438, "sampling/importance_sampling_ratio/min": 0.30937132239341736, "sampling/sampling_logp_difference/max": 1.173213005065918, "sampling/sampling_logp_difference/mean": 0.028799764811992645, "step": 1516 }, { "clip_ratio/high_max": 0.027737777214497328, "clip_ratio/high_mean": 0.027737777214497328, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.027737777214497328, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 57.75, "completions/mean_terminated_length": 57.75, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.31292806193232536, "epoch": 0.060931035867775235, "frac_reward_zero_std": 0.0, "grad_norm": 2.4998230934143066, "learning_rate": 5.406060606060607e-06, "loss": -0.015, "num_tokens": 3413722.0, "reward": 0.9230245351791382, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9230245351791382, "reward_meter_std": 0.1628594845533371, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1628594696521759, "reward_total_composite_mean": 0.9230245351791382, "reward_total_composite_std": 0.1628594845533371, "reward_total_mean": 0.9230245351791382, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9230245351791382, "rewards/meter/std": 0.1628594845533371, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9230245351791382, "rewards/total_composite/std": 0.1628594845533371, "sampling/importance_sampling_ratio/max": 1.4794626235961914, "sampling/importance_sampling_ratio/mean": 1.0011502504348755, "sampling/importance_sampling_ratio/min": 0.28013360500335693, "sampling/sampling_logp_difference/max": 1.2724885940551758, "sampling/sampling_logp_difference/mean": 0.04387671872973442, "step": 1517 }, { "clip_ratio/high_max": 0.04732227721251547, "clip_ratio/high_mean": 0.04732227721251547, "clip_ratio/low_mean": 0.030917395371943712, "clip_ratio/low_min": 0.030917395371943712, "clip_ratio/region_mean": 0.07823967258445919, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 60.625, "completions/mean_terminated_length": 60.625, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.6950614899396896, "epoch": 0.06097120134956019, "frac_reward_zero_std": 0.0, "grad_norm": 6.170195579528809, "learning_rate": 5.4030303030303036e-06, "loss": 0.0093, "num_tokens": 3415375.0, "reward": 0.5306375622749329, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5306375622749329, "reward_meter_std": 0.289710134267807, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.289710134267807, "reward_total_composite_mean": 0.5306375622749329, "reward_total_composite_std": 0.289710134267807, "reward_total_mean": 0.5306375622749329, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5306375622749329, "rewards/meter/std": 0.289710134267807, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5306375622749329, "rewards/total_composite/std": 0.289710134267807, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0169787406921387, "sampling/importance_sampling_ratio/min": 0.24199077486991882, "sampling/sampling_logp_difference/max": 1.4188556671142578, "sampling/sampling_logp_difference/mean": 0.07036992907524109, "step": 1518 }, { "clip_ratio/high_max": 0.028688129736110568, "clip_ratio/high_mean": 0.028688129736110568, "clip_ratio/low_mean": 0.020800751633942127, "clip_ratio/low_min": 0.020800751633942127, "clip_ratio/region_mean": 0.049488881370052695, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 290.875, "completions/mean_terminated_length": 290.875, "completions/min_length": 274.0, "completions/min_terminated_length": 274.0, "entropy": 0.7886361554265022, "epoch": 0.06101136683134514, "frac_reward_zero_std": 0.0, "grad_norm": 3.1494743824005127, "learning_rate": 5.400000000000001e-06, "loss": -0.0293, "num_tokens": 3419414.0, "reward": 0.5596551299095154, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.7045454978942871, "reward_count_adherence_std": 0.04208274558186531, "reward_meter_mean": 0.9781413078308105, "reward_meter_std": 0.051752228289842606, "reward_repeat_penalty_mean": 0.9208333492279053, "reward_repeat_penalty_std": 0.12346728891134262, "reward_std": 0.2521139085292816, "reward_total_composite_mean": 0.5596551299095154, "reward_total_composite_std": 0.2521139085292816, "reward_total_mean": 0.5596551299095154, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.7045454978942871, "rewards/count_adherence/std": 0.04208274558186531, "rewards/meter/mean": 0.9781413078308105, "rewards/meter/std": 0.051752228289842606, "rewards/repeat_penalty/mean": 0.9208333492279053, "rewards/repeat_penalty/std": 0.12346728891134262, "rewards/total_composite/mean": 0.5596551299095154, "rewards/total_composite/std": 0.2521139085292816, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0232716798782349, "sampling/importance_sampling_ratio/min": 0.19335635006427765, "sampling/sampling_logp_difference/max": 1.6432204246520996, "sampling/sampling_logp_difference/mean": 0.07325157523155212, "step": 1519 }, { "clip_ratio/high_max": 0.004295865655876696, "clip_ratio/high_mean": 0.004295865655876696, "clip_ratio/low_mean": 0.0028572361916303635, "clip_ratio/low_min": 0.0028572361916303635, "clip_ratio/region_mean": 0.00715310184750706, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 88.625, "completions/mean_terminated_length": 88.625, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.1767467763274908, "epoch": 0.0610515323131301, "frac_reward_zero_std": 0.0, "grad_norm": 2.657116413116455, "learning_rate": 5.396969696969697e-06, "loss": 0.0163, "num_tokens": 3421603.0, "reward": 0.8182051777839661, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9171510934829712, "reward_meter_std": 0.1984940767288208, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.1850285530090332, "reward_total_composite_mean": 0.8182051777839661, "reward_total_composite_std": 0.1850285530090332, "reward_total_mean": 0.8182051777839661, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9171510934829712, "rewards/meter/std": 0.1984940767288208, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8182051777839661, "rewards/total_composite/std": 0.1850285530090332, "sampling/importance_sampling_ratio/max": 1.6496562957763672, "sampling/importance_sampling_ratio/mean": 1.0104413032531738, "sampling/importance_sampling_ratio/min": 0.46796390414237976, "sampling/sampling_logp_difference/max": 0.759364128112793, "sampling/sampling_logp_difference/mean": 0.01846635341644287, "step": 1520 }, { "clip_ratio/high_max": 0.010897728730924428, "clip_ratio/high_mean": 0.010897728730924428, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.010897728730924428, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.5, "completions/mean_terminated_length": 68.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.09944538865238428, "epoch": 0.06109169779491505, "frac_reward_zero_std": 0.0, "grad_norm": 1.7797563076019287, "learning_rate": 5.3939393939393945e-06, "loss": 0.0033, "num_tokens": 3423423.0, "reward": 0.9934801459312439, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9934801459312439, "reward_meter_std": 0.0036428512539714575, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00364283611997962, "reward_total_composite_mean": 0.9934801459312439, "reward_total_composite_std": 0.0036428512539714575, "reward_total_mean": 0.9934801459312439, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9934801459312439, "rewards/meter/std": 0.0036428512539714575, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934801459312439, "rewards/total_composite/std": 0.0036428512539714575, "sampling/importance_sampling_ratio/max": 1.2535090446472168, "sampling/importance_sampling_ratio/mean": 1.0017439126968384, "sampling/importance_sampling_ratio/min": 0.14572004973888397, "sampling/sampling_logp_difference/max": 1.9260679483413696, "sampling/sampling_logp_difference/mean": 0.01598731242120266, "step": 1521 }, { "clip_ratio/high_max": 0.019181564333848655, "clip_ratio/high_mean": 0.019181564333848655, "clip_ratio/low_mean": 0.0058302809484303, "clip_ratio/low_min": 0.0058302809484303, "clip_ratio/region_mean": 0.025011845282278955, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.37458037585020065, "epoch": 0.061131863276700005, "frac_reward_zero_std": 0.0, "grad_norm": 7.3915534019470215, "learning_rate": 5.390909090909091e-06, "loss": -0.0022, "num_tokens": 3425367.0, "reward": 0.9976125955581665, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976125955581665, "reward_meter_std": 0.0013877107994630933, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0013877045130357146, "reward_total_composite_mean": 0.9976125955581665, "reward_total_composite_std": 0.0013877107994630933, "reward_total_mean": 0.9976125955581665, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976125955581665, "rewards/meter/std": 0.0013877107994630933, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976125955581665, "rewards/total_composite/std": 0.0013877107994630933, "sampling/importance_sampling_ratio/max": 1.817367672920227, "sampling/importance_sampling_ratio/mean": 1.0058274269104004, "sampling/importance_sampling_ratio/min": 0.5116690397262573, "sampling/sampling_logp_difference/max": 0.6700773239135742, "sampling/sampling_logp_difference/mean": 0.040052663534879684, "step": 1522 }, { "clip_ratio/high_max": 0.04335171659477055, "clip_ratio/high_mean": 0.04335171659477055, "clip_ratio/low_mean": 0.010714286006987095, "clip_ratio/low_min": 0.010714286006987095, "clip_ratio/region_mean": 0.054066002601757646, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.5, "completions/mean_terminated_length": 34.5, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.7306949496269226, "epoch": 0.06117202875848496, "frac_reward_zero_std": 0.0, "grad_norm": 9.1239013671875, "learning_rate": 5.387878787878789e-06, "loss": 0.0106, "num_tokens": 3426883.0, "reward": 0.9722611904144287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9722611904144287, "reward_meter_std": 0.0555478073656559, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05554782226681709, "reward_total_composite_mean": 0.9722611904144287, "reward_total_composite_std": 0.0555478073656559, "reward_total_mean": 0.9722611904144287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9722611904144287, "rewards/meter/std": 0.0555478073656559, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9722611904144287, "rewards/total_composite/std": 0.0555478073656559, "sampling/importance_sampling_ratio/max": 1.5298467874526978, "sampling/importance_sampling_ratio/mean": 1.0279422998428345, "sampling/importance_sampling_ratio/min": 0.3437480628490448, "sampling/sampling_logp_difference/max": 1.0678462982177734, "sampling/sampling_logp_difference/mean": 0.07421485334634781, "step": 1523 }, { "clip_ratio/high_max": 0.03156230039894581, "clip_ratio/high_mean": 0.03156230039894581, "clip_ratio/low_mean": 0.01814199541695416, "clip_ratio/low_min": 0.01814199541695416, "clip_ratio/region_mean": 0.04970429581589997, "completions/clipped_ratio": 0.0, "completions/max_length": 113.0, "completions/max_terminated_length": 113.0, "completions/mean_length": 108.75, "completions/mean_terminated_length": 108.75, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.894000805914402, "epoch": 0.06121219424026991, "frac_reward_zero_std": 0.0, "grad_norm": 5.130848407745361, "learning_rate": 5.384848484848485e-06, "loss": 0.0095, "num_tokens": 3429089.0, "reward": 0.9915084838867188, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9915084838867188, "reward_meter_std": 0.007808292284607887, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007808296009898186, "reward_total_composite_mean": 0.9915084838867188, "reward_total_composite_std": 0.007808292284607887, "reward_total_mean": 0.9915084838867188, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9915084838867188, "rewards/meter/std": 0.007808292284607887, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9915084838867188, "rewards/total_composite/std": 0.007808292284607887, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0207182168960571, "sampling/importance_sampling_ratio/min": 0.19624315202236176, "sampling/sampling_logp_difference/max": 1.6284008026123047, "sampling/sampling_logp_difference/mean": 0.08183825016021729, "step": 1524 }, { "clip_ratio/high_max": 0.012915466912090778, "clip_ratio/high_mean": 0.012915466912090778, "clip_ratio/low_mean": 0.0028735632076859474, "clip_ratio/low_min": 0.0028735632076859474, "clip_ratio/region_mean": 0.015789030119776726, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 86.875, "completions/mean_terminated_length": 86.875, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.09439879935234785, "epoch": 0.06125235972205487, "frac_reward_zero_std": 0.0, "grad_norm": 2.808112382888794, "learning_rate": 5.381818181818183e-06, "loss": 0.0026, "num_tokens": 3431040.0, "reward": 0.9431794881820679, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99290531873703, "reward_meter_std": 0.0034807121846824884, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09114135801792145, "reward_total_composite_mean": 0.9431794881820679, "reward_total_composite_std": 0.09114136546850204, "reward_total_mean": 0.9431794881820679, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99290531873703, "rewards/meter/std": 0.0034807121846824884, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9431794881820679, "rewards/total_composite/std": 0.09114136546850204, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0029078722000122, "sampling/importance_sampling_ratio/min": 0.48398464918136597, "sampling/sampling_logp_difference/max": 0.7286503314971924, "sampling/sampling_logp_difference/mean": 0.017489619553089142, "step": 1525 }, { "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/low_mean": 0.017182219657115638, "clip_ratio/low_min": 0.017182219657115638, "clip_ratio/region_mean": 0.0197332400130108, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 95.125, "completions/mean_terminated_length": 95.125, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.13843786902725697, "epoch": 0.06129252520383982, "frac_reward_zero_std": 0.0, "grad_norm": 2.34173321723938, "learning_rate": 5.378787878787879e-06, "loss": -0.0074, "num_tokens": 3433169.0, "reward": 0.8195037841796875, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9933592081069946, "reward_meter_std": 0.00307702855207026, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07000675052404404, "reward_total_composite_mean": 0.8195037841796875, "reward_total_composite_std": 0.07000674307346344, "reward_total_mean": 0.8195037841796875, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9933592081069946, "rewards/meter/std": 0.00307702855207026, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8195037841796875, "rewards/total_composite/std": 0.07000674307346344, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006750464439392, "sampling/importance_sampling_ratio/min": 0.2805590033531189, "sampling/sampling_logp_difference/max": 1.2709712982177734, "sampling/sampling_logp_difference/mean": 0.023438630625605583, "step": 1526 }, { "clip_ratio/high_max": 0.044742210768163204, "clip_ratio/high_mean": 0.044742210768163204, "clip_ratio/low_mean": 0.006133304443210363, "clip_ratio/low_min": 0.006133304443210363, "clip_ratio/region_mean": 0.05087551521137357, "completions/clipped_ratio": 0.0, "completions/max_length": 149.0, "completions/max_terminated_length": 149.0, "completions/mean_length": 142.75, "completions/mean_terminated_length": 142.75, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.7129531875252724, "epoch": 0.061332690685624774, "frac_reward_zero_std": 0.0, "grad_norm": 3.2660653591156006, "learning_rate": 5.375757575757576e-06, "loss": 0.0164, "num_tokens": 3435703.0, "reward": 0.9411816596984863, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994535505771637, "reward_meter_std": 0.0031923020724207163, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.10628911107778549, "reward_std": 0.10509292781352997, "reward_total_composite_mean": 0.9411816596984863, "reward_total_composite_std": 0.10509292781352997, "reward_total_mean": 0.9411816596984863, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994535505771637, "rewards/meter/std": 0.0031923020724207163, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.9411816596984863, "rewards/total_composite/std": 0.10509292781352997, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.02262282371521, "sampling/importance_sampling_ratio/min": 0.19429361820220947, "sampling/sampling_logp_difference/max": 1.6383848190307617, "sampling/sampling_logp_difference/mean": 0.07457228004932404, "step": 1527 }, { "clip_ratio/high_max": 0.02779570873826742, "clip_ratio/high_mean": 0.02779570873826742, "clip_ratio/low_mean": 0.009738856926560402, "clip_ratio/low_min": 0.009738856926560402, "clip_ratio/region_mean": 0.037534565664827824, "completions/clipped_ratio": 0.0, "completions/max_length": 97.0, "completions/max_terminated_length": 97.0, "completions/mean_length": 93.375, "completions/mean_terminated_length": 93.375, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.5462022610008717, "epoch": 0.06137285616740973, "frac_reward_zero_std": 0.0, "grad_norm": 5.586496353149414, "learning_rate": 5.372727272727273e-06, "loss": -0.0161, "num_tokens": 3437794.0, "reward": 0.8450676202774048, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8450676202774048, "reward_meter_std": 0.220209538936615, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2202095091342926, "reward_total_composite_mean": 0.8450676202774048, "reward_total_composite_std": 0.220209538936615, "reward_total_mean": 0.8450676202774048, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8450676202774048, "rewards/meter/std": 0.220209538936615, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8450676202774048, "rewards/total_composite/std": 0.220209538936615, "sampling/importance_sampling_ratio/max": 1.8447113037109375, "sampling/importance_sampling_ratio/mean": 1.0161728858947754, "sampling/importance_sampling_ratio/min": 0.37451109290122986, "sampling/sampling_logp_difference/max": 0.9821338653564453, "sampling/sampling_logp_difference/mean": 0.05679401755332947, "step": 1528 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 33.0, "completions/mean_terminated_length": 33.0, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.017181317321956158, "epoch": 0.06141302164919468, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.36969696969697e-06, "loss": 0.0, "num_tokens": 3439074.0, "reward": 0.9988001585006714, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988001585006714, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9988001585006714, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9988001585006714, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988001585006714, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988001585006714, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.045628309249878, "sampling/importance_sampling_ratio/mean": 1.0021947622299194, "sampling/importance_sampling_ratio/min": 0.9977096915245056, "sampling/sampling_logp_difference/max": 0.04461796581745148, "sampling/sampling_logp_difference/mean": 0.0021949801594018936, "step": 1529 }, { "clip_ratio/high_max": 0.015155904809944332, "clip_ratio/high_mean": 0.015155904809944332, "clip_ratio/low_mean": 0.003791360300965607, "clip_ratio/low_min": 0.003791360300965607, "clip_ratio/region_mean": 0.01894726511090994, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.1264043264091015, "epoch": 0.061453187130979636, "frac_reward_zero_std": 0.0, "grad_norm": 3.0035605430603027, "learning_rate": 5.366666666666666e-06, "loss": 0.0132, "num_tokens": 3440826.0, "reward": 0.9936773180961609, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9936773180961609, "reward_meter_std": 0.0019825000781565905, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001982480986043811, "reward_total_composite_mean": 0.9936773180961609, "reward_total_composite_std": 0.0019825000781565905, "reward_total_mean": 0.9936773180961609, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9936773180961609, "rewards/meter/std": 0.0019825000781565905, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9936773180961609, "rewards/total_composite/std": 0.0019825000781565905, "sampling/importance_sampling_ratio/max": 1.562909722328186, "sampling/importance_sampling_ratio/mean": 1.0015543699264526, "sampling/importance_sampling_ratio/min": 0.4409589171409607, "sampling/sampling_logp_difference/max": 0.8188035488128662, "sampling/sampling_logp_difference/mean": 0.02317051589488983, "step": 1530 }, { "clip_ratio/high_max": 0.045307138469070196, "clip_ratio/high_mean": 0.045307138469070196, "clip_ratio/low_mean": 0.007166183087974787, "clip_ratio/low_min": 0.007166183087974787, "clip_ratio/region_mean": 0.05247332155704498, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 106.875, "completions/mean_terminated_length": 106.875, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.799922339618206, "epoch": 0.06149335261276459, "frac_reward_zero_std": 0.0, "grad_norm": 5.306570053100586, "learning_rate": 5.3636363636363645e-06, "loss": -0.0098, "num_tokens": 3443097.0, "reward": 0.8351579904556274, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.980573296546936, "reward_meter_std": 0.03288979455828667, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.3450130224227905, "reward_total_composite_mean": 0.8351579904556274, "reward_total_composite_std": 0.3450130224227905, "reward_total_mean": 0.8351579904556274, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.980573296546936, "rewards/meter/std": 0.03288979455828667, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8351579904556274, "rewards/total_composite/std": 0.3450130224227905, "sampling/importance_sampling_ratio/max": 1.9533883333206177, "sampling/importance_sampling_ratio/mean": 1.011287808418274, "sampling/importance_sampling_ratio/min": 0.13037173449993134, "sampling/sampling_logp_difference/max": 2.037365436553955, "sampling/sampling_logp_difference/mean": 0.08104944229125977, "step": 1531 }, { "clip_ratio/high_max": 0.035965551156550646, "clip_ratio/high_mean": 0.035965551156550646, "clip_ratio/low_mean": 0.01080156397074461, "clip_ratio/low_min": 0.01080156397074461, "clip_ratio/region_mean": 0.046767115127295256, "completions/clipped_ratio": 0.0, "completions/max_length": 165.0, "completions/max_terminated_length": 165.0, "completions/mean_length": 160.5, "completions/mean_terminated_length": 160.5, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.5020967051386833, "epoch": 0.061533518094549544, "frac_reward_zero_std": 0.0, "grad_norm": 3.583285331726074, "learning_rate": 5.360606060606061e-06, "loss": -0.0044, "num_tokens": 3445821.0, "reward": 0.7505040168762207, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8937799334526062, "reward_meter_std": 0.1616331785917282, "reward_repeat_penalty_mean": 0.8333333134651184, "reward_repeat_penalty_std": 0.11878276616334915, "reward_std": 0.18041202425956726, "reward_total_composite_mean": 0.7505040168762207, "reward_total_composite_std": 0.18041202425956726, "reward_total_mean": 0.7505040168762207, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8937799334526062, "rewards/meter/std": 0.1616331785917282, "rewards/repeat_penalty/mean": 0.8333333134651184, "rewards/repeat_penalty/std": 0.11878276616334915, "rewards/total_composite/mean": 0.7505040168762207, "rewards/total_composite/std": 0.18041202425956726, "sampling/importance_sampling_ratio/max": 1.947861909866333, "sampling/importance_sampling_ratio/mean": 1.012855052947998, "sampling/importance_sampling_ratio/min": 0.2624739110469818, "sampling/sampling_logp_difference/max": 1.3376035690307617, "sampling/sampling_logp_difference/mean": 0.06113876402378082, "step": 1532 }, { "clip_ratio/high_max": 0.010924369795247912, "clip_ratio/high_mean": 0.010924369795247912, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.014712248696014285, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.375, "completions/mean_terminated_length": 34.375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.13886073511093855, "epoch": 0.0615736835763345, "frac_reward_zero_std": 0.0, "grad_norm": 4.299396514892578, "learning_rate": 5.357575757575758e-06, "loss": 0.0012, "num_tokens": 3447376.0, "reward": 0.9954771995544434, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9954771995544434, "reward_meter_std": 0.002512240083888173, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00251222332008183, "reward_total_composite_mean": 0.9954771995544434, "reward_total_composite_std": 0.002512240083888173, "reward_total_mean": 0.9954771995544434, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9954771995544434, "rewards/meter/std": 0.002512240083888173, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9954771995544434, "rewards/total_composite/std": 0.002512240083888173, "sampling/importance_sampling_ratio/max": 1.5883126258850098, "sampling/importance_sampling_ratio/mean": 1.009590983390808, "sampling/importance_sampling_ratio/min": 0.7345414757728577, "sampling/sampling_logp_difference/max": 0.46267223358154297, "sampling/sampling_logp_difference/mean": 0.01558676641434431, "step": 1533 }, { "clip_ratio/high_max": 0.023747874423861504, "clip_ratio/high_mean": 0.023747874423861504, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.023747874423861504, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 50.0, "completions/mean_length": 106.625, "completions/mean_terminated_length": 48.71428680419922, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.16696601919829845, "epoch": 0.06161384905811945, "frac_reward_zero_std": 0.0, "grad_norm": 1.0674489736557007, "learning_rate": 5.3545454545454546e-06, "loss": -0.1403, "num_tokens": 3449061.0, "reward": 0.80141282081604, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8021861910820007, "reward_meter_std": 0.325016587972641, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.32718130946159363, "reward_total_composite_mean": 0.80141282081604, "reward_total_composite_std": 0.327181339263916, "reward_total_mean": 0.80141282081604, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8021861910820007, "rewards/meter/std": 0.325016587972641, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.80141282081604, "rewards/total_composite/std": 0.327181339263916, "sampling/importance_sampling_ratio/max": 1.8218151330947876, "sampling/importance_sampling_ratio/mean": 1.0069955587387085, "sampling/importance_sampling_ratio/min": 0.08425161987543106, "sampling/sampling_logp_difference/max": 2.473947525024414, "sampling/sampling_logp_difference/mean": 0.038483839482069016, "step": 1534 }, { "clip_ratio/high_max": 0.0292887045070529, "clip_ratio/high_mean": 0.0292887045070529, "clip_ratio/low_mean": 0.018939394503831863, "clip_ratio/low_min": 0.018939394503831863, "clip_ratio/region_mean": 0.04822809901088476, "completions/clipped_ratio": 0.0, "completions/max_length": 174.0, "completions/max_terminated_length": 174.0, "completions/mean_length": 153.0, "completions/mean_terminated_length": 153.0, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.5641567148268223, "epoch": 0.061654014539904406, "frac_reward_zero_std": 0.0, "grad_norm": 4.45563268661499, "learning_rate": 5.351515151515152e-06, "loss": -0.0071, "num_tokens": 3451653.0, "reward": 0.6905776858329773, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_meter_mean": 0.7749242782592773, "reward_meter_std": 0.3511018455028534, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.07856741547584534, "reward_std": 0.3377683758735657, "reward_total_composite_mean": 0.6905776858329773, "reward_total_composite_std": 0.3377683758735657, "reward_total_mean": 0.6905776858329773, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/meter/mean": 0.7749242782592773, "rewards/meter/std": 0.3511018455028534, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.6905776858329773, "rewards/total_composite/std": 0.3377683758735657, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0109103918075562, "sampling/importance_sampling_ratio/min": 0.2659173607826233, "sampling/sampling_logp_difference/max": 1.3245697021484375, "sampling/sampling_logp_difference/mean": 0.05948156490921974, "step": 1535 }, { "clip_ratio/high_max": 0.03292302007321268, "clip_ratio/high_mean": 0.03292302007321268, "clip_ratio/low_mean": 0.016904115676879883, "clip_ratio/low_min": 0.016904115676879883, "clip_ratio/region_mean": 0.049827135750092566, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.4786563478410244, "epoch": 0.06169418002168936, "frac_reward_zero_std": 0.0, "grad_norm": 4.13173770904541, "learning_rate": 5.348484848484848e-06, "loss": -0.0035, "num_tokens": 3453309.0, "reward": 0.8998743295669556, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8998743295669556, "reward_meter_std": 0.1591634452342987, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1591634303331375, "reward_total_composite_mean": 0.8998743295669556, "reward_total_composite_std": 0.1591634452342987, "reward_total_mean": 0.8998743295669556, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8998743295669556, "rewards/meter/std": 0.1591634452342987, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8998743295669556, "rewards/total_composite/std": 0.1591634452342987, "sampling/importance_sampling_ratio/max": 1.8585302829742432, "sampling/importance_sampling_ratio/mean": 1.0069003105163574, "sampling/importance_sampling_ratio/min": 0.35935643315315247, "sampling/sampling_logp_difference/max": 1.0234405994415283, "sampling/sampling_logp_difference/mean": 0.05023188889026642, "step": 1536 }, { "clip_ratio/high_max": 0.01667593652382493, "clip_ratio/high_mean": 0.01667593652382493, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.02058218652382493, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 30.5, "completions/mean_terminated_length": 30.5, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.229267543181777, "epoch": 0.061734345503474314, "frac_reward_zero_std": 0.0, "grad_norm": 8.413439750671387, "learning_rate": 5.3454545454545455e-06, "loss": 0.0252, "num_tokens": 3454809.0, "reward": 0.9885121583938599, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9885121583938599, "reward_meter_std": 0.006495221517980099, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006495220120996237, "reward_total_composite_mean": 0.9885121583938599, "reward_total_composite_std": 0.006495221517980099, "reward_total_mean": 0.9885121583938599, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9885121583938599, "rewards/meter/std": 0.006495221517980099, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9885121583938599, "rewards/total_composite/std": 0.006495221517980099, "sampling/importance_sampling_ratio/max": 1.6943132877349854, "sampling/importance_sampling_ratio/mean": 1.0021849870681763, "sampling/importance_sampling_ratio/min": 0.40363138914108276, "sampling/sampling_logp_difference/max": 0.9072532653808594, "sampling/sampling_logp_difference/mean": 0.03234010562300682, "step": 1537 }, { "clip_ratio/high_max": 0.018546971026808023, "clip_ratio/high_mean": 0.018546971026808023, "clip_ratio/low_mean": 0.007634902372956276, "clip_ratio/low_min": 0.007634902372956276, "clip_ratio/region_mean": 0.0261818733997643, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.3516472205519676, "epoch": 0.06177451098525927, "frac_reward_zero_std": 0.0, "grad_norm": 3.6924545764923096, "learning_rate": 5.342424242424244e-06, "loss": -0.0059, "num_tokens": 3456577.0, "reward": 0.9983944892883301, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983944892883301, "reward_meter_std": 0.0004069809801876545, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00040697952499613166, "reward_total_composite_mean": 0.9983944892883301, "reward_total_composite_std": 0.0004069809801876545, "reward_total_mean": 0.9983944892883301, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983944892883301, "rewards/meter/std": 0.0004069809801876545, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983944892883301, "rewards/total_composite/std": 0.0004069809801876545, "sampling/importance_sampling_ratio/max": 1.5474638938903809, "sampling/importance_sampling_ratio/mean": 1.0129139423370361, "sampling/importance_sampling_ratio/min": 0.3473750650882721, "sampling/sampling_logp_difference/max": 1.0573501586914062, "sampling/sampling_logp_difference/mean": 0.0381365530192852, "step": 1538 }, { "clip_ratio/high_max": 0.009995099389925599, "clip_ratio/high_mean": 0.009995099389925599, "clip_ratio/low_mean": 0.0014367816038429737, "clip_ratio/low_min": 0.0014367816038429737, "clip_ratio/region_mean": 0.011431880993768573, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 87.125, "completions/mean_terminated_length": 87.125, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.08502828422933817, "epoch": 0.06181467646704422, "frac_reward_zero_std": 0.0, "grad_norm": 2.03717303276062, "learning_rate": 5.33939393939394e-06, "loss": -0.0048, "num_tokens": 3458618.0, "reward": 0.9949196577072144, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949196577072144, "reward_meter_std": 0.00045097857946529984, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00045097683323547244, "reward_total_composite_mean": 0.9949196577072144, "reward_total_composite_std": 0.00045097857946529984, "reward_total_mean": 0.9949196577072144, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949196577072144, "rewards/meter/std": 0.00045097857946529984, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949196577072144, "rewards/total_composite/std": 0.00045097857946529984, "sampling/importance_sampling_ratio/max": 1.4961625337600708, "sampling/importance_sampling_ratio/mean": 1.0028852224349976, "sampling/importance_sampling_ratio/min": 0.3457634150981903, "sampling/sampling_logp_difference/max": 1.0620005130767822, "sampling/sampling_logp_difference/mean": 0.0125720901414752, "step": 1539 }, { "clip_ratio/high_max": 0.023029743460938334, "clip_ratio/high_mean": 0.023029743460938334, "clip_ratio/low_mean": 0.009146341122686863, "clip_ratio/low_min": 0.009146341122686863, "clip_ratio/region_mean": 0.0321760845836252, "completions/clipped_ratio": 0.0, "completions/max_length": 50.0, "completions/max_terminated_length": 50.0, "completions/mean_length": 48.375, "completions/mean_terminated_length": 48.375, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.2700903480872512, "epoch": 0.061854841948829176, "frac_reward_zero_std": 0.0, "grad_norm": 12.248023986816406, "learning_rate": 5.336363636363637e-06, "loss": -0.0382, "num_tokens": 3460405.0, "reward": 0.8395511507987976, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8395511507987976, "reward_meter_std": 0.27874672412872314, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.27874669432640076, "reward_total_composite_mean": 0.8395511507987976, "reward_total_composite_std": 0.27874672412872314, "reward_total_mean": 0.8395511507987976, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8395511507987976, "rewards/meter/std": 0.27874672412872314, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8395511507987976, "rewards/total_composite/std": 0.27874672412872314, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0040042400360107, "sampling/importance_sampling_ratio/min": 0.35930031538009644, "sampling/sampling_logp_difference/max": 1.4850921630859375, "sampling/sampling_logp_difference/mean": 0.04395786300301552, "step": 1540 }, { "clip_ratio/high_max": 0.0310292961075902, "clip_ratio/high_mean": 0.0310292961075902, "clip_ratio/low_mean": 0.008196720853447914, "clip_ratio/low_min": 0.008196720853447914, "clip_ratio/region_mean": 0.03922601696103811, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 60.0, "completions/mean_terminated_length": 60.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.47805059514939785, "epoch": 0.06189500743061413, "frac_reward_zero_std": 0.0, "grad_norm": 7.992240905761719, "learning_rate": 5.333333333333334e-06, "loss": 0.0077, "num_tokens": 3462213.0, "reward": 0.9441971778869629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9441971778869629, "reward_meter_std": 0.13217569887638092, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.13217571377754211, "reward_total_composite_mean": 0.9441971778869629, "reward_total_composite_std": 0.13217569887638092, "reward_total_mean": 0.9441971778869629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9441971778869629, "rewards/meter/std": 0.13217569887638092, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9441971778869629, "rewards/total_composite/std": 0.13217569887638092, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0145553350448608, "sampling/importance_sampling_ratio/min": 0.21890106797218323, "sampling/sampling_logp_difference/max": 1.5191354751586914, "sampling/sampling_logp_difference/mean": 0.06318628787994385, "step": 1541 }, { "clip_ratio/high_max": 0.04202882433310151, "clip_ratio/high_mean": 0.04202882433310151, "clip_ratio/low_mean": 0.005937984678894281, "clip_ratio/low_min": 0.005937984678894281, "clip_ratio/region_mean": 0.04796680901199579, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 127.625, "completions/mean_terminated_length": 127.625, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.6336969211697578, "epoch": 0.061935172912399084, "frac_reward_zero_std": 0.0, "grad_norm": 3.4401650428771973, "learning_rate": 5.330303030303031e-06, "loss": 0.0082, "num_tokens": 3464650.0, "reward": 0.9360376596450806, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9891485571861267, "reward_meter_std": 0.004043546039611101, "reward_repeat_penalty_mean": 0.9464285373687744, "reward_repeat_penalty_std": 0.10628911107778549, "reward_std": 0.10430650413036346, "reward_total_composite_mean": 0.9360376596450806, "reward_total_composite_std": 0.10430651903152466, "reward_total_mean": 0.9360376596450806, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9891485571861267, "rewards/meter/std": 0.004043546039611101, "rewards/repeat_penalty/mean": 0.9464285373687744, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.9360376596450806, "rewards/total_composite/std": 0.10430651903152466, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0177416801452637, "sampling/importance_sampling_ratio/min": 0.19583091139793396, "sampling/sampling_logp_difference/max": 1.6305036544799805, "sampling/sampling_logp_difference/mean": 0.07292276620864868, "step": 1542 }, { "clip_ratio/high_max": 0.009101442527025938, "clip_ratio/high_mean": 0.009101442527025938, "clip_ratio/low_mean": 0.00649872503709048, "clip_ratio/low_min": 0.00649872503709048, "clip_ratio/region_mean": 0.015600167564116418, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 96.125, "completions/mean_terminated_length": 96.125, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.08730749879032373, "epoch": 0.06197533839418404, "frac_reward_zero_std": 0.0, "grad_norm": 5.007582664489746, "learning_rate": 5.327272727272727e-06, "loss": -0.0036, "num_tokens": 3466723.0, "reward": 0.7870810627937317, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7870810627937317, "reward_meter_std": 0.3985043168067932, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3985043168067932, "reward_total_composite_mean": 0.7870810627937317, "reward_total_composite_std": 0.3985043168067932, "reward_total_mean": 0.7870810627937317, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7870810627937317, "rewards/meter/std": 0.3985043168067932, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7870810627937317, "rewards/total_composite/std": 0.3985043168067932, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9980475306510925, "sampling/importance_sampling_ratio/min": 0.20322878658771515, "sampling/sampling_logp_difference/max": 1.5934228897094727, "sampling/sampling_logp_difference/mean": 0.026287445798516273, "step": 1543 }, { "clip_ratio/high_max": 0.01856476883403957, "clip_ratio/high_mean": 0.01856476883403957, "clip_ratio/low_mean": 0.007875503972172737, "clip_ratio/low_min": 0.007875503972172737, "clip_ratio/region_mean": 0.026440272806212306, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.14403609465807676, "epoch": 0.06201550387596899, "frac_reward_zero_std": 0.0, "grad_norm": 4.138071060180664, "learning_rate": 5.324242424242425e-06, "loss": -0.0189, "num_tokens": 3468547.0, "reward": 0.9653145670890808, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9653145670890808, "reward_meter_std": 0.03330826759338379, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03330826386809349, "reward_total_composite_mean": 0.9653145670890808, "reward_total_composite_std": 0.03330826759338379, "reward_total_mean": 0.9653145670890808, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9653145670890808, "rewards/meter/std": 0.03330826759338379, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9653145670890808, "rewards/total_composite/std": 0.03330826759338379, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0086570978164673, "sampling/importance_sampling_ratio/min": 0.4773949682712555, "sampling/sampling_logp_difference/max": 0.9563794136047363, "sampling/sampling_logp_difference/mean": 0.025457728654146194, "step": 1544 }, { "clip_ratio/high_max": 0.02582784171681851, "clip_ratio/high_mean": 0.02582784171681851, "clip_ratio/low_mean": 0.01173020526766777, "clip_ratio/low_min": 0.01173020526766777, "clip_ratio/region_mean": 0.03755804698448628, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.25, "completions/mean_terminated_length": 67.25, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.33554743882268667, "epoch": 0.062055669357753945, "frac_reward_zero_std": 0.0, "grad_norm": 5.288153648376465, "learning_rate": 5.321212121212122e-06, "loss": -0.0078, "num_tokens": 3470357.0, "reward": 0.9962025284767151, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962025284767151, "reward_meter_std": 0.00486297020688653, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004862963687628508, "reward_total_composite_mean": 0.9962025284767151, "reward_total_composite_std": 0.00486297020688653, "reward_total_mean": 0.9962025284767151, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962025284767151, "rewards/meter/std": 0.00486297020688653, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9962025284767151, "rewards/total_composite/std": 0.00486297020688653, "sampling/importance_sampling_ratio/max": 1.475104570388794, "sampling/importance_sampling_ratio/mean": 1.0034021139144897, "sampling/importance_sampling_ratio/min": 0.2659450173377991, "sampling/sampling_logp_difference/max": 1.3244657516479492, "sampling/sampling_logp_difference/mean": 0.038305725902318954, "step": 1545 }, { "clip_ratio/high_max": 0.013565883273258805, "clip_ratio/high_mean": 0.013565883273258805, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/region_mean": 0.019425258273258805, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 127.75, "completions/mean_terminated_length": 127.75, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.14278420340269804, "epoch": 0.0620958348395389, "frac_reward_zero_std": 0.0, "grad_norm": 2.919037342071533, "learning_rate": 5.318181818181819e-06, "loss": -0.0044, "num_tokens": 3472779.0, "reward": 0.6911537647247314, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9427036046981812, "reward_meter_std": 0.14980448782444, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.12467768788337708, "reward_total_composite_mean": 0.6911537647247314, "reward_total_composite_std": 0.12467769533395767, "reward_total_mean": 0.6911537647247314, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9427036046981812, "rewards/meter/std": 0.14980448782444, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.6911537647247314, "rewards/total_composite/std": 0.12467769533395767, "sampling/importance_sampling_ratio/max": 1.6036888360977173, "sampling/importance_sampling_ratio/mean": 1.0017569065093994, "sampling/importance_sampling_ratio/min": 0.13969002664089203, "sampling/sampling_logp_difference/max": 1.9683294296264648, "sampling/sampling_logp_difference/mean": 0.022355277091264725, "step": 1546 }, { "clip_ratio/high_max": 0.04056679271161556, "clip_ratio/high_mean": 0.04056679271161556, "clip_ratio/low_mean": 0.015021929983049631, "clip_ratio/low_min": 0.015021929983049631, "clip_ratio/region_mean": 0.055588722694665194, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 60.5, "completions/mean_terminated_length": 60.5, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.7294313032180071, "epoch": 0.06213600032132385, "frac_reward_zero_std": 0.0, "grad_norm": 6.456394195556641, "learning_rate": 5.3151515151515155e-06, "loss": -0.0066, "num_tokens": 3474543.0, "reward": 0.6942555904388428, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.796026885509491, "reward_meter_std": 0.2907305955886841, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4039342701435089, "reward_total_composite_mean": 0.6942555904388428, "reward_total_composite_std": 0.4039342701435089, "reward_total_mean": 0.6942555904388428, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.796026885509491, "rewards/meter/std": 0.2907305955886841, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6942555904388428, "rewards/total_composite/std": 0.4039342701435089, "sampling/importance_sampling_ratio/max": 1.528890609741211, "sampling/importance_sampling_ratio/mean": 1.0131072998046875, "sampling/importance_sampling_ratio/min": 0.26189157366752625, "sampling/sampling_logp_difference/max": 1.3398246765136719, "sampling/sampling_logp_difference/mean": 0.07712767273187637, "step": 1547 }, { "clip_ratio/high_max": 0.03996537090279162, "clip_ratio/high_mean": 0.03996537090279162, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/region_mean": 0.047541128704324365, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 34.25, "completions/mean_terminated_length": 34.25, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.7250464260578156, "epoch": 0.06217616580310881, "frac_reward_zero_std": 0.0, "grad_norm": 11.082610130310059, "learning_rate": 5.312121212121213e-06, "loss": 0.0213, "num_tokens": 3476009.0, "reward": 0.8179029822349548, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8179029822349548, "reward_meter_std": 0.24513599276542664, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.24513597786426544, "reward_total_composite_mean": 0.8179029822349548, "reward_total_composite_std": 0.24513599276542664, "reward_total_mean": 0.8179029822349548, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8179029822349548, "rewards/meter/std": 0.24513599276542664, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8179029822349548, "rewards/total_composite/std": 0.24513599276542664, "sampling/importance_sampling_ratio/max": 1.7788945436477661, "sampling/importance_sampling_ratio/mean": 1.0309749841690063, "sampling/importance_sampling_ratio/min": 0.35903966426849365, "sampling/sampling_logp_difference/max": 1.024322509765625, "sampling/sampling_logp_difference/mean": 0.07286211103200912, "step": 1548 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 30.0, "completions/mean_terminated_length": 30.0, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.06372858118265867, "epoch": 0.06221633128489376, "frac_reward_zero_std": 0.0, "grad_norm": 8.7092924118042, "learning_rate": 5.309090909090909e-06, "loss": -0.0072, "num_tokens": 3477433.0, "reward": 0.9894814491271973, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9894814491271973, "reward_meter_std": 0.00789269246160984, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007892684079706669, "reward_total_composite_mean": 0.9894814491271973, "reward_total_composite_std": 0.00789269246160984, "reward_total_mean": 0.9894814491271973, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9894814491271973, "rewards/meter/std": 0.00789269246160984, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9894814491271973, "rewards/total_composite/std": 0.00789269246160984, "sampling/importance_sampling_ratio/max": 1.0616073608398438, "sampling/importance_sampling_ratio/mean": 0.9964293241500854, "sampling/importance_sampling_ratio/min": 0.29690131545066833, "sampling/sampling_logp_difference/max": 1.21435546875, "sampling/sampling_logp_difference/mean": 0.014411688782274723, "step": 1549 }, { "clip_ratio/high_max": 0.010820218129083514, "clip_ratio/high_mean": 0.010820218129083514, "clip_ratio/low_mean": 0.004273816477507353, "clip_ratio/low_min": 0.004273816477507353, "clip_ratio/region_mean": 0.015094034606590867, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 58.25, "completions/mean_terminated_length": 58.25, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.09011476766318083, "epoch": 0.062256496766678715, "frac_reward_zero_std": 0.0, "grad_norm": 4.305907249450684, "learning_rate": 5.306060606060606e-06, "loss": 0.0078, "num_tokens": 3479115.0, "reward": 0.7832661867141724, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9904994964599609, "reward_meter_std": 0.009607957676053047, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.17251639068126678, "reward_std": 0.1659567803144455, "reward_total_composite_mean": 0.7832661867141724, "reward_total_composite_std": 0.1659567952156067, "reward_total_mean": 0.7832661867141724, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9904994964599609, "rewards/meter/std": 0.009607957676053047, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.17251639068126678, "rewards/total_composite/mean": 0.7832661867141724, "rewards/total_composite/std": 0.1659567952156067, "sampling/importance_sampling_ratio/max": 1.4585957527160645, "sampling/importance_sampling_ratio/mean": 1.000774621963501, "sampling/importance_sampling_ratio/min": 0.40299662947654724, "sampling/sampling_logp_difference/max": 0.9088270664215088, "sampling/sampling_logp_difference/mean": 0.01842333748936653, "step": 1550 }, { "epoch": 0.062256496766678715, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 299.7692307692308, "eval_completions/max_terminated_length": 299.7692307692308, "eval_completions/mean_length": 174.25, "eval_completions/mean_terminated_length": 174.25, "eval_completions/min_length": 62.30769230769231, "eval_completions/min_terminated_length": 62.30769230769231, "eval_entropy": 0.3527725820357983, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 3479115.0, "eval_reward": 0.5367342164883246, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.8626743646768423, "eval_reward_count_adherence_std": 0.15295347571372986, "eval_reward_meter_mean": 0.7260088416246268, "eval_reward_meter_std": 0.3893573456085645, "eval_reward_repeat_penalty_mean": 0.8659125016285822, "eval_reward_repeat_penalty_std": 0.14054490625858307, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5367342164883246, "eval_reward_total_composite_std": 0.3409125472490604, "eval_reward_total_mean": 0.5367342164883246, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.8626743646768423, "eval_rewards/count_adherence/std": 0.15295347571372986, "eval_rewards/meter/mean": 0.7260088416246268, "eval_rewards/meter/std": 0.3893573456085645, "eval_rewards/repeat_penalty/mean": 0.8659125016285822, "eval_rewards/repeat_penalty/std": 0.14054490625858307, "eval_rewards/total_composite/mean": 0.5367342164883246, "eval_rewards/total_composite/std": 0.3409125472490604, "eval_runtime": 59.4299, "eval_samples_per_second": 1.75, "eval_sampling/importance_sampling_ratio/max": 1.5303159401966975, "eval_sampling/importance_sampling_ratio/mean": 1.0097044522945697, "eval_sampling/importance_sampling_ratio/min": 0.34075071719976574, "eval_sampling/sampling_logp_difference/max": 1.1005325317382812, "eval_sampling/sampling_logp_difference/mean": 0.032458307078251473, "eval_steps_per_second": 0.219, "step": 1550 }, { "clip_ratio/high_max": 0.012758038472384214, "clip_ratio/high_mean": 0.012758038472384214, "clip_ratio/low_mean": 0.006647313362918794, "clip_ratio/low_min": 0.006647313362918794, "clip_ratio/region_mean": 0.01940535183530301, "completions/clipped_ratio": 0.0, "completions/max_length": 174.0, "completions/max_terminated_length": 174.0, "completions/mean_length": 168.125, "completions/mean_terminated_length": 168.125, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.14000887237489223, "epoch": 0.06229666224846367, "frac_reward_zero_std": 0.0, "grad_norm": 1.714144229888916, "learning_rate": 5.303030303030303e-06, "loss": 0.0162, "num_tokens": 3481876.0, "reward": 0.7904214859008789, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984342455863953, "reward_meter_std": 0.00023047976719681174, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.0927247703075409, "reward_std": 0.09253095090389252, "reward_total_composite_mean": 0.7904214859008789, "reward_total_composite_std": 0.09253095835447311, "reward_total_mean": 0.7904214859008789, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984342455863953, "rewards/meter/std": 0.00023047976719681174, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.0927247703075409, "rewards/total_composite/mean": 0.7904214859008789, "rewards/total_composite/std": 0.09253095835447311, "sampling/importance_sampling_ratio/max": 1.999273419380188, "sampling/importance_sampling_ratio/mean": 1.002803921699524, "sampling/importance_sampling_ratio/min": 0.12313804030418396, "sampling/sampling_logp_difference/max": 2.094449281692505, "sampling/sampling_logp_difference/mean": 0.02464020438492298, "step": 1551 }, { "clip_ratio/high_max": 0.026524552376940846, "clip_ratio/high_mean": 0.026524552376940846, "clip_ratio/low_mean": 0.009651749860495329, "clip_ratio/low_min": 0.009651749860495329, "clip_ratio/region_mean": 0.036176302237436175, "completions/clipped_ratio": 0.0, "completions/max_length": 223.0, "completions/max_terminated_length": 223.0, "completions/mean_length": 212.875, "completions/mean_terminated_length": 212.875, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "entropy": 0.4306017607450485, "epoch": 0.06233682773024862, "frac_reward_zero_std": 0.0, "grad_norm": 2.4818339347839355, "learning_rate": 5.300000000000001e-06, "loss": -0.01, "num_tokens": 3485227.0, "reward": 0.7158641219139099, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.828125, "reward_count_adherence_std": 0.06469365209341049, "reward_meter_mean": 0.9701088666915894, "reward_meter_std": 0.02329552173614502, "reward_repeat_penalty_mean": 0.8910256624221802, "reward_repeat_penalty_std": 0.1113983765244484, "reward_std": 0.10737311094999313, "reward_total_composite_mean": 0.7158641219139099, "reward_total_composite_std": 0.10737311840057373, "reward_total_mean": 0.7158641219139099, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.828125, "rewards/count_adherence/std": 0.06469365209341049, "rewards/meter/mean": 0.9701088666915894, "rewards/meter/std": 0.02329552173614502, "rewards/repeat_penalty/mean": 0.8910256624221802, "rewards/repeat_penalty/std": 0.1113983765244484, "rewards/total_composite/mean": 0.7158641219139099, "rewards/total_composite/std": 0.10737311840057373, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0085928440093994, "sampling/importance_sampling_ratio/min": 0.21455803513526917, "sampling/sampling_logp_difference/max": 1.539175033569336, "sampling/sampling_logp_difference/mean": 0.044115372002124786, "step": 1552 }, { "clip_ratio/high_max": 0.022814685944467783, "clip_ratio/high_mean": 0.022814685944467783, "clip_ratio/low_mean": 0.0174350953893736, "clip_ratio/low_min": 0.0174350953893736, "clip_ratio/region_mean": 0.04024978133384138, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 65.375, "completions/mean_terminated_length": 65.375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.40439430996775627, "epoch": 0.06237699321203358, "frac_reward_zero_std": 0.0, "grad_norm": 7.115604400634766, "learning_rate": 5.296969696969697e-06, "loss": -0.0033, "num_tokens": 3487126.0, "reward": 0.977942705154419, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.977942705154419, "reward_meter_std": 0.007237757556140423, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007237738464027643, "reward_total_composite_mean": 0.977942705154419, "reward_total_composite_std": 0.007237757556140423, "reward_total_mean": 0.977942705154419, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.977942705154419, "rewards/meter/std": 0.007237757556140423, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.977942705154419, "rewards/total_composite/std": 0.007237757556140423, "sampling/importance_sampling_ratio/max": 1.691331386566162, "sampling/importance_sampling_ratio/mean": 1.0108517408370972, "sampling/importance_sampling_ratio/min": 0.40712234377861023, "sampling/sampling_logp_difference/max": 0.8986415863037109, "sampling/sampling_logp_difference/mean": 0.03986657038331032, "step": 1553 }, { "clip_ratio/high_max": 0.004815331194549799, "clip_ratio/high_mean": 0.004815331194549799, "clip_ratio/low_mean": 0.0021819902467541397, "clip_ratio/low_min": 0.0021819902467541397, "clip_ratio/region_mean": 0.006997321441303939, "completions/clipped_ratio": 0.0, "completions/max_length": 240.0, "completions/max_terminated_length": 240.0, "completions/mean_length": 229.125, "completions/mean_terminated_length": 229.125, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.0460505741648376, "epoch": 0.06241715869381853, "frac_reward_zero_std": 0.0, "grad_norm": 2.9659626483917236, "learning_rate": 5.293939393939395e-06, "loss": -0.0326, "num_tokens": 3490583.0, "reward": 0.6974078416824341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.859375, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9962524771690369, "reward_meter_std": 0.002178115537390113, "reward_repeat_penalty_mean": 0.8135302066802979, "reward_repeat_penalty_std": 0.05918560549616814, "reward_std": 0.07142962515354156, "reward_total_composite_mean": 0.6974078416824341, "reward_total_composite_std": 0.07142963260412216, "reward_total_mean": 0.6974078416824341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.859375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9962524771690369, "rewards/meter/std": 0.002178115537390113, "rewards/repeat_penalty/mean": 0.8135302066802979, "rewards/repeat_penalty/std": 0.05918560549616814, "rewards/total_composite/mean": 0.6974078416824341, "rewards/total_composite/std": 0.07142963260412216, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000474452972412, "sampling/importance_sampling_ratio/min": 0.059884749352931976, "sampling/sampling_logp_difference/max": 2.815333366394043, "sampling/sampling_logp_difference/mean": 0.013559546321630478, "step": 1554 }, { "clip_ratio/high_max": 0.00838999031111598, "clip_ratio/high_mean": 0.00838999031111598, "clip_ratio/low_mean": 0.0032157512614503503, "clip_ratio/low_min": 0.0032157512614503503, "clip_ratio/region_mean": 0.01160574157256633, "completions/clipped_ratio": 0.0, "completions/max_length": 204.0, "completions/max_terminated_length": 204.0, "completions/mean_length": 193.125, "completions/mean_terminated_length": 193.125, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.05383288115262985, "epoch": 0.062457324175603485, "frac_reward_zero_std": 0.0, "grad_norm": 4.632443428039551, "learning_rate": 5.290909090909091e-06, "loss": 0.0094, "num_tokens": 3493632.0, "reward": 0.6953567266464233, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984503984451294, "reward_meter_std": 0.00041957091889344156, "reward_repeat_penalty_mean": 0.8125, "reward_repeat_penalty_std": 0.08256731927394867, "reward_std": 0.0707344338297844, "reward_total_composite_mean": 0.6953567266464233, "reward_total_composite_std": 0.0707344338297844, "reward_total_mean": 0.6953567266464233, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984503984451294, "rewards/meter/std": 0.00041957091889344156, "rewards/repeat_penalty/mean": 0.8125, "rewards/repeat_penalty/std": 0.08256731927394867, "rewards/total_composite/mean": 0.6953567266464233, "rewards/total_composite/std": 0.0707344338297844, "sampling/importance_sampling_ratio/max": 1.899730920791626, "sampling/importance_sampling_ratio/mean": 0.9994859099388123, "sampling/importance_sampling_ratio/min": 0.18284127116203308, "sampling/sampling_logp_difference/max": 1.6991368532180786, "sampling/sampling_logp_difference/mean": 0.015556903555989265, "step": 1555 }, { "clip_ratio/high_max": 0.014115164172835648, "clip_ratio/high_mean": 0.014115164172835648, "clip_ratio/low_mean": 0.010138889076188207, "clip_ratio/low_min": 0.010138889076188207, "clip_ratio/region_mean": 0.024254053249023855, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 71.625, "completions/mean_terminated_length": 71.625, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.11559192929416895, "epoch": 0.06249748965738844, "frac_reward_zero_std": 0.0, "grad_norm": 3.2562649250030518, "learning_rate": 5.287878787878788e-06, "loss": 0.0201, "num_tokens": 3495501.0, "reward": 0.9960213303565979, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9960213303565979, "reward_meter_std": 0.0024177224840968847, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0024177224840968847, "reward_total_composite_mean": 0.9960213303565979, "reward_total_composite_std": 0.0024177224840968847, "reward_total_mean": 0.9960213303565979, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9960213303565979, "rewards/meter/std": 0.0024177224840968847, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9960213303565979, "rewards/total_composite/std": 0.0024177224840968847, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0080082416534424, "sampling/importance_sampling_ratio/min": 0.27203330397605896, "sampling/sampling_logp_difference/max": 1.301830768585205, "sampling/sampling_logp_difference/mean": 0.02655681222677231, "step": 1556 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.0036764706019312143, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.06252996809780598, "epoch": 0.06253765513917339, "frac_reward_zero_std": 0.0, "grad_norm": 1.289771556854248, "learning_rate": 5.284848484848485e-06, "loss": -0.0016, "num_tokens": 3497301.0, "reward": 0.9974825382232666, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974825382232666, "reward_meter_std": 0.0009680635412223637, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000968057313002646, "reward_total_composite_mean": 0.9974825382232666, "reward_total_composite_std": 0.0009680635412223637, "reward_total_mean": 0.9974825382232666, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974825382232666, "rewards/meter/std": 0.0009680635412223637, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974825382232666, "rewards/total_composite/std": 0.0009680635412223637, "sampling/importance_sampling_ratio/max": 1.1250584125518799, "sampling/importance_sampling_ratio/mean": 1.0034716129302979, "sampling/importance_sampling_ratio/min": 0.7130227088928223, "sampling/sampling_logp_difference/max": 0.3382420539855957, "sampling/sampling_logp_difference/mean": 0.007437488529831171, "step": 1557 }, { "clip_ratio/high_max": 0.010714285774156451, "clip_ratio/high_mean": 0.010714285774156451, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.010714285774156451, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 35.25, "completions/mean_terminated_length": 35.25, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.06664726650342345, "epoch": 0.06257782062095835, "frac_reward_zero_std": 0.0, "grad_norm": 1.997252345085144, "learning_rate": 5.281818181818183e-06, "loss": 0.0063, "num_tokens": 3498783.0, "reward": 0.9977306127548218, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977306127548218, "reward_meter_std": 4.97752262162976e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.97752262162976e-05, "reward_total_composite_mean": 0.9977306127548218, "reward_total_composite_std": 4.97752262162976e-05, "reward_total_mean": 0.9977306127548218, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977306127548218, "rewards/meter/std": 4.97752262162976e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977306127548218, "rewards/total_composite/std": 4.97752262162976e-05, "sampling/importance_sampling_ratio/max": 1.3929730653762817, "sampling/importance_sampling_ratio/mean": 1.0048854351043701, "sampling/importance_sampling_ratio/min": 0.7421995401382446, "sampling/sampling_logp_difference/max": 0.33144044876098633, "sampling/sampling_logp_difference/mean": 0.00860452838242054, "step": 1558 }, { "clip_ratio/high_max": 0.0341619870159775, "clip_ratio/high_mean": 0.0341619870159775, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/region_mean": 0.03617811598815024, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.0, "completions/mean_terminated_length": 62.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.4067396577447653, "epoch": 0.0626179861027433, "frac_reward_zero_std": 0.0, "grad_norm": 2.993905782699585, "learning_rate": 5.278787878787879e-06, "loss": -0.0004, "num_tokens": 3500559.0, "reward": 0.9484682083129883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9898895025253296, "reward_meter_std": 0.006041497457772493, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11559665948152542, "reward_total_composite_mean": 0.9484682083129883, "reward_total_composite_std": 0.11559666693210602, "reward_total_mean": 0.9484682083129883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9898895025253296, "rewards/meter/std": 0.006041497457772493, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9484682083129883, "rewards/total_composite/std": 0.11559666693210602, "sampling/importance_sampling_ratio/max": 1.9443280696868896, "sampling/importance_sampling_ratio/mean": 1.0192768573760986, "sampling/importance_sampling_ratio/min": 0.37098902463912964, "sampling/sampling_logp_difference/max": 0.9915828704833984, "sampling/sampling_logp_difference/mean": 0.0489659458398819, "step": 1559 }, { "clip_ratio/high_max": 0.030295601580291986, "clip_ratio/high_mean": 0.030295601580291986, "clip_ratio/low_mean": 0.020772642455995083, "clip_ratio/low_min": 0.020772642455995083, "clip_ratio/region_mean": 0.05106824403628707, "completions/clipped_ratio": 0.0, "completions/max_length": 177.0, "completions/max_terminated_length": 177.0, "completions/mean_length": 164.5, "completions/mean_terminated_length": 164.5, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.6054803691804409, "epoch": 0.06265815158452825, "frac_reward_zero_std": 0.0, "grad_norm": 3.2149598598480225, "learning_rate": 5.2757575757575764e-06, "loss": 0.0004, "num_tokens": 3503371.0, "reward": 0.9209266901016235, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9614261388778687, "reward_meter_std": 0.03620919957756996, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.05824670195579529, "reward_total_composite_mean": 0.9209266901016235, "reward_total_composite_std": 0.058246713131666183, "reward_total_mean": 0.9209266901016235, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9614261388778687, "rewards/meter/std": 0.03620919957756996, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9209266901016235, "rewards/total_composite/std": 0.058246713131666183, "sampling/importance_sampling_ratio/max": 1.8964616060256958, "sampling/importance_sampling_ratio/mean": 1.010610580444336, "sampling/importance_sampling_ratio/min": 0.15101857483386993, "sampling/sampling_logp_difference/max": 1.890352487564087, "sampling/sampling_logp_difference/mean": 0.06021958962082863, "step": 1560 }, { "clip_ratio/high_max": 0.01949126087129116, "clip_ratio/high_mean": 0.01949126087129116, "clip_ratio/low_mean": 0.026302700862288475, "clip_ratio/low_min": 0.026302700862288475, "clip_ratio/region_mean": 0.045793961733579636, "completions/clipped_ratio": 0.0, "completions/max_length": 180.0, "completions/max_terminated_length": 180.0, "completions/mean_length": 172.25, "completions/mean_terminated_length": 172.25, "completions/min_length": 161.0, "completions/min_terminated_length": 161.0, "entropy": 0.5600994266569614, "epoch": 0.06269831706631321, "frac_reward_zero_std": 0.0, "grad_norm": 3.249744176864624, "learning_rate": 5.272727272727273e-06, "loss": -0.0154, "num_tokens": 3506277.0, "reward": 0.9116497039794922, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9807374477386475, "reward_meter_std": 0.04032743349671364, "reward_repeat_penalty_mean": 0.9305555820465088, "reward_repeat_penalty_std": 0.08266931027173996, "reward_std": 0.07851860672235489, "reward_total_composite_mean": 0.9116497039794922, "reward_total_composite_std": 0.07851860672235489, "reward_total_mean": 0.9116497039794922, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9807374477386475, "rewards/meter/std": 0.04032743349671364, "rewards/repeat_penalty/mean": 0.9305555820465088, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.9116497039794922, "rewards/total_composite/std": 0.07851860672235489, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0178204774856567, "sampling/importance_sampling_ratio/min": 0.11868584156036377, "sampling/sampling_logp_difference/max": 2.131275177001953, "sampling/sampling_logp_difference/mean": 0.06819657236337662, "step": 1561 }, { "clip_ratio/high_max": 0.017139121424406767, "clip_ratio/high_mean": 0.017139121424406767, "clip_ratio/low_mean": 0.016098485328257084, "clip_ratio/low_min": 0.016098485328257084, "clip_ratio/region_mean": 0.03323760675266385, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.4796396978199482, "epoch": 0.06273848254809816, "frac_reward_zero_std": 0.0, "grad_norm": 7.787070274353027, "learning_rate": 5.26969696969697e-06, "loss": 0.0416, "num_tokens": 3507984.0, "reward": 0.9753642678260803, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9753642678260803, "reward_meter_std": 0.012682100757956505, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.012682083994150162, "reward_total_composite_mean": 0.9753642678260803, "reward_total_composite_std": 0.012682100757956505, "reward_total_mean": 0.9753642678260803, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9753642678260803, "rewards/meter/std": 0.012682100757956505, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9753642678260803, "rewards/total_composite/std": 0.012682100757956505, "sampling/importance_sampling_ratio/max": 1.8774487972259521, "sampling/importance_sampling_ratio/mean": 1.020971417427063, "sampling/importance_sampling_ratio/min": 0.4774416387081146, "sampling/sampling_logp_difference/max": 0.7393133640289307, "sampling/sampling_logp_difference/mean": 0.05309029668569565, "step": 1562 }, { "clip_ratio/high_max": 0.0008680555620230734, "clip_ratio/high_mean": 0.0008680555620230734, "clip_ratio/low_mean": 0.012661085231229663, "clip_ratio/low_min": 0.012661085231229663, "clip_ratio/region_mean": 0.013529140793252736, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 137.375, "completions/mean_terminated_length": 137.375, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.15207071043550968, "epoch": 0.06277864802988312, "frac_reward_zero_std": 0.0, "grad_norm": 2.302091121673584, "learning_rate": 5.2666666666666665e-06, "loss": -0.0149, "num_tokens": 3510443.0, "reward": 0.7311152815818787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985888004302979, "reward_meter_std": 0.0002428983716527, "reward_repeat_penalty_mean": 0.7321428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05054369196295738, "reward_total_composite_mean": 0.7311152815818787, "reward_total_composite_std": 0.05054369568824768, "reward_total_mean": 0.7311152815818787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985888004302979, "rewards/meter/std": 0.0002428983716527, "rewards/repeat_penalty/mean": 0.7321428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.7311152815818787, "rewards/total_composite/std": 0.05054369568824768, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0065100193023682, "sampling/importance_sampling_ratio/min": 0.3813832104206085, "sampling/sampling_logp_difference/max": 0.9639506340026855, "sampling/sampling_logp_difference/mean": 0.01936904713511467, "step": 1563 }, { "clip_ratio/high_max": 0.033232659101486206, "clip_ratio/high_mean": 0.033232659101486206, "clip_ratio/low_mean": 0.02909582480788231, "clip_ratio/low_min": 0.02909582480788231, "clip_ratio/region_mean": 0.062328483909368515, "completions/clipped_ratio": 0.0, "completions/max_length": 139.0, "completions/max_terminated_length": 139.0, "completions/mean_length": 131.75, "completions/mean_terminated_length": 131.75, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.7511504292488098, "epoch": 0.06281881351166807, "frac_reward_zero_std": 0.0, "grad_norm": 5.161999702453613, "learning_rate": 5.263636363636364e-06, "loss": -0.0126, "num_tokens": 3512785.0, "reward": 0.692800760269165, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.718358039855957, "reward_meter_std": 0.32531455159187317, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.32817986607551575, "reward_total_composite_mean": 0.692800760269165, "reward_total_composite_std": 0.32817986607551575, "reward_total_mean": 0.692800760269165, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.718358039855957, "rewards/meter/std": 0.32531455159187317, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.692800760269165, "rewards/total_composite/std": 0.32817986607551575, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01358962059021, "sampling/importance_sampling_ratio/min": 0.3377377986907959, "sampling/sampling_logp_difference/max": 1.0854854583740234, "sampling/sampling_logp_difference/mean": 0.07563211023807526, "step": 1564 }, { "clip_ratio/high_max": 0.041696065571159124, "clip_ratio/high_mean": 0.041696065571159124, "clip_ratio/low_mean": 0.025564054027199745, "clip_ratio/low_min": 0.025564054027199745, "clip_ratio/region_mean": 0.06726011959835887, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.7624497786164284, "epoch": 0.06285897899345302, "frac_reward_zero_std": 0.0, "grad_norm": 6.407342433929443, "learning_rate": 5.26060606060606e-06, "loss": -0.0189, "num_tokens": 3514505.0, "reward": 0.6359885931015015, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6359885931015015, "reward_meter_std": 0.39930614829063416, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.39930611848831177, "reward_total_composite_mean": 0.6359885931015015, "reward_total_composite_std": 0.39930614829063416, "reward_total_mean": 0.6359885931015015, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6359885931015015, "rewards/meter/std": 0.39930614829063416, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6359885931015015, "rewards/total_composite/std": 0.39930614829063416, "sampling/importance_sampling_ratio/max": 1.9221627712249756, "sampling/importance_sampling_ratio/mean": 1.0197231769561768, "sampling/importance_sampling_ratio/min": 0.20447883009910583, "sampling/sampling_logp_difference/max": 1.5872907638549805, "sampling/sampling_logp_difference/mean": 0.07870090752840042, "step": 1565 }, { "clip_ratio/high_max": 0.05206052586436272, "clip_ratio/high_mean": 0.05206052586436272, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/region_mean": 0.057417668867856264, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.6091113425791264, "epoch": 0.06289914447523798, "frac_reward_zero_std": 0.0, "grad_norm": 4.360117435455322, "learning_rate": 5.257575757575758e-06, "loss": -0.0005, "num_tokens": 3516356.0, "reward": 0.9724572896957397, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9724572896957397, "reward_meter_std": 0.05403033271431923, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.054030340164899826, "reward_total_composite_mean": 0.9724572896957397, "reward_total_composite_std": 0.05403033271431923, "reward_total_mean": 0.9724572896957397, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9724572896957397, "rewards/meter/std": 0.05403033271431923, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9724572896957397, "rewards/total_composite/std": 0.05403033271431923, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0121489763259888, "sampling/importance_sampling_ratio/min": 0.18474335968494415, "sampling/sampling_logp_difference/max": 1.6887876987457275, "sampling/sampling_logp_difference/mean": 0.06206238269805908, "step": 1566 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.007247899193316698, "clip_ratio/low_min": 0.007247899193316698, "clip_ratio/region_mean": 0.007247899193316698, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.5, "completions/mean_terminated_length": 68.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.0569656016305089, "epoch": 0.06293930995702293, "frac_reward_zero_std": 0.0, "grad_norm": 0.27344831824302673, "learning_rate": 5.2545454545454555e-06, "loss": 0.0001, "num_tokens": 3518144.0, "reward": 0.9978020191192627, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978020191192627, "reward_meter_std": 2.9081666070851497e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.9075883503537625e-05, "reward_total_composite_mean": 0.9978020191192627, "reward_total_composite_std": 2.9081666070851497e-05, "reward_total_mean": 0.9978020191192627, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978020191192627, "rewards/meter/std": 2.9081666070851497e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978020191192627, "rewards/total_composite/std": 2.9081666070851497e-05, "sampling/importance_sampling_ratio/max": 1.357559084892273, "sampling/importance_sampling_ratio/mean": 1.0018134117126465, "sampling/importance_sampling_ratio/min": 0.47649744153022766, "sampling/sampling_logp_difference/max": 0.7412929534912109, "sampling/sampling_logp_difference/mean": 0.010215016081929207, "step": 1567 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 100.125, "completions/mean_terminated_length": 100.125, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.04269820963963866, "epoch": 0.06297947543880789, "frac_reward_zero_std": 0.0, "grad_norm": 0.3887041211128235, "learning_rate": 5.251515151515152e-06, "loss": -0.0006, "num_tokens": 3520369.0, "reward": 0.9977670311927795, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977670311927795, "reward_meter_std": 4.242736758897081e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.243615330778994e-05, "reward_total_composite_mean": 0.9977670311927795, "reward_total_composite_std": 4.242736758897081e-05, "reward_total_mean": 0.9977670311927795, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977670311927795, "rewards/meter/std": 4.242736758897081e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977670311927795, "rewards/total_composite/std": 4.242736758897081e-05, "sampling/importance_sampling_ratio/max": 1.1633974313735962, "sampling/importance_sampling_ratio/mean": 1.00063157081604, "sampling/importance_sampling_ratio/min": 0.47633635997772217, "sampling/sampling_logp_difference/max": 0.741631031036377, "sampling/sampling_logp_difference/mean": 0.007036585360765457, "step": 1568 }, { "clip_ratio/high_max": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/region_mean": 0.003969253972172737, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 63.125, "completions/mean_terminated_length": 63.125, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.033813337329775095, "epoch": 0.06301964092059284, "frac_reward_zero_std": 0.0, "grad_norm": 19.87078857421875, "learning_rate": 5.248484848484849e-06, "loss": 0.0094, "num_tokens": 3522282.0, "reward": 0.998197078704834, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998197078704834, "reward_meter_std": 0.00027716331533156335, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00027716669137589633, "reward_total_composite_mean": 0.998197078704834, "reward_total_composite_std": 0.00027716331533156335, "reward_total_mean": 0.998197078704834, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998197078704834, "rewards/meter/std": 0.00027716331533156335, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998197078704834, "rewards/total_composite/std": 0.00027716331533156335, "sampling/importance_sampling_ratio/max": 1.3447296619415283, "sampling/importance_sampling_ratio/mean": 0.9992884993553162, "sampling/importance_sampling_ratio/min": 0.40890324115753174, "sampling/sampling_logp_difference/max": 0.8942767381668091, "sampling/sampling_logp_difference/mean": 0.010409246198832989, "step": 1569 }, { "clip_ratio/high_max": 0.04102045390754938, "clip_ratio/high_mean": 0.04102045390754938, "clip_ratio/low_mean": 0.028282265178859234, "clip_ratio/low_min": 0.028282265178859234, "clip_ratio/region_mean": 0.06930271908640862, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 65.125, "completions/mean_terminated_length": 65.125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.8175845369696617, "epoch": 0.0630598064023778, "frac_reward_zero_std": 0.0, "grad_norm": 6.926992893218994, "learning_rate": 5.245454545454546e-06, "loss": 0.0363, "num_tokens": 3524067.0, "reward": 0.6135989427566528, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7359851598739624, "reward_meter_std": 0.4039470851421356, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.46367475390434265, "reward_total_composite_mean": 0.6135989427566528, "reward_total_composite_std": 0.46367478370666504, "reward_total_mean": 0.6135989427566528, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7359851598739624, "rewards/meter/std": 0.4039470851421356, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6135989427566528, "rewards/total_composite/std": 0.46367478370666504, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0079141855239868, "sampling/importance_sampling_ratio/min": 0.29272621870040894, "sampling/sampling_logp_difference/max": 1.2285175323486328, "sampling/sampling_logp_difference/mean": 0.08760727196931839, "step": 1570 }, { "clip_ratio/high_max": 0.011825977824628353, "clip_ratio/high_mean": 0.011825977824628353, "clip_ratio/low_mean": 0.006336958380416036, "clip_ratio/low_min": 0.006336958380416036, "clip_ratio/region_mean": 0.01816293620504439, "completions/clipped_ratio": 0.0, "completions/max_length": 185.0, "completions/max_terminated_length": 185.0, "completions/mean_length": 171.125, "completions/mean_terminated_length": 171.125, "completions/min_length": 168.0, "completions/min_terminated_length": 168.0, "entropy": 0.0667440458200872, "epoch": 0.06309997188416275, "frac_reward_zero_std": 0.0, "grad_norm": 2.949753999710083, "learning_rate": 5.242424242424244e-06, "loss": 0.0237, "num_tokens": 3527020.0, "reward": 0.767343282699585, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.9974609613418579, "reward_meter_std": 0.0009353406494483352, "reward_repeat_penalty_mean": 0.7897727489471436, "reward_repeat_penalty_std": 0.03682740405201912, "reward_std": 0.05765299126505852, "reward_total_composite_mean": 0.767343282699585, "reward_total_composite_std": 0.05765299126505852, "reward_total_mean": 0.767343282699585, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.9974609613418579, "rewards/meter/std": 0.0009353406494483352, "rewards/repeat_penalty/mean": 0.7897727489471436, "rewards/repeat_penalty/std": 0.03682740405201912, "rewards/total_composite/mean": 0.767343282699585, "rewards/total_composite/std": 0.05765299126505852, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9989144206047058, "sampling/importance_sampling_ratio/min": 0.46636128425598145, "sampling/sampling_logp_difference/max": 1.0565416812896729, "sampling/sampling_logp_difference/mean": 0.014616935513913631, "step": 1571 }, { "clip_ratio/high_max": 0.012186205829493701, "clip_ratio/high_mean": 0.012186205829493701, "clip_ratio/low_mean": 0.015551732387393713, "clip_ratio/low_min": 0.015551732387393713, "clip_ratio/region_mean": 0.027737938216887414, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 297.0, "completions/mean_length": 308.25, "completions/mean_terminated_length": 279.14288330078125, "completions/min_length": 265.0, "completions/min_terminated_length": 265.0, "entropy": 0.41537686437368393, "epoch": 0.0631401373659477, "frac_reward_zero_std": 0.0, "grad_norm": 1.639418125152588, "learning_rate": 5.23939393939394e-06, "loss": -0.218, "num_tokens": 3530638.0, "reward": 0.5252456665039062, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.862500011920929, "reward_count_adherence_std": 0.0517548993229866, "reward_meter_mean": 0.6877518892288208, "reward_meter_std": 0.3425425887107849, "reward_repeat_penalty_mean": 0.9172793626785278, "reward_repeat_penalty_std": 0.07193652540445328, "reward_std": 0.28777357935905457, "reward_total_composite_mean": 0.5252456665039062, "reward_total_composite_std": 0.28777360916137695, "reward_total_mean": 0.5252456665039062, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.862500011920929, "rewards/count_adherence/std": 0.0517548993229866, "rewards/meter/mean": 0.6877518892288208, "rewards/meter/std": 0.3425425887107849, "rewards/repeat_penalty/mean": 0.9172793626785278, "rewards/repeat_penalty/std": 0.07193652540445328, "rewards/total_composite/mean": 0.5252456665039062, "rewards/total_composite/std": 0.28777360916137695, "sampling/importance_sampling_ratio/max": 1.8714102506637573, "sampling/importance_sampling_ratio/mean": 1.012639045715332, "sampling/importance_sampling_ratio/min": 0.08456631004810333, "sampling/sampling_logp_difference/max": 2.470219373703003, "sampling/sampling_logp_difference/mean": 0.0495113804936409, "step": 1572 }, { "clip_ratio/high_max": 0.047869643196463585, "clip_ratio/high_mean": 0.047869643196463585, "clip_ratio/low_mean": 0.033473065122962, "clip_ratio/low_min": 0.033473065122962, "clip_ratio/region_mean": 0.08134270831942558, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 64.375, "completions/mean_terminated_length": 64.375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.7125434167683125, "epoch": 0.06318030284773266, "frac_reward_zero_std": 0.0, "grad_norm": 5.333371162414551, "learning_rate": 5.236363636363637e-06, "loss": 0.0007, "num_tokens": 3532465.0, "reward": 0.6559607982635498, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6559607982635498, "reward_meter_std": 0.3895518183708191, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3895518183708191, "reward_total_composite_mean": 0.6559607982635498, "reward_total_composite_std": 0.3895518183708191, "reward_total_mean": 0.6559607982635498, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6559607982635498, "rewards/meter/std": 0.3895518183708191, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6559607982635498, "rewards/total_composite/std": 0.3895518183708191, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0150277614593506, "sampling/importance_sampling_ratio/min": 0.15392273664474487, "sampling/sampling_logp_difference/max": 1.8713045120239258, "sampling/sampling_logp_difference/mean": 0.07227914035320282, "step": 1573 }, { "clip_ratio/high_max": 0.04302619933150709, "clip_ratio/high_mean": 0.04302619933150709, "clip_ratio/low_mean": 0.00818392145447433, "clip_ratio/low_min": 0.00818392145447433, "clip_ratio/region_mean": 0.05121012078598142, "completions/clipped_ratio": 0.0, "completions/max_length": 100.0, "completions/max_terminated_length": 100.0, "completions/mean_length": 94.125, "completions/mean_terminated_length": 94.125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.8282850012183189, "epoch": 0.06322046832951761, "frac_reward_zero_std": 0.0, "grad_norm": 6.877980709075928, "learning_rate": 5.233333333333334e-06, "loss": -0.0041, "num_tokens": 3534610.0, "reward": 0.8243577480316162, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8490567207336426, "reward_meter_std": 0.342186838388443, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.3378320336341858, "reward_total_composite_mean": 0.8243577480316162, "reward_total_composite_std": 0.3378320336341858, "reward_total_mean": 0.8243577480316162, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8490567207336426, "rewards/meter/std": 0.342186838388443, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8243577480316162, "rewards/total_composite/std": 0.3378320336341858, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.00845468044281, "sampling/importance_sampling_ratio/min": 0.10528901219367981, "sampling/sampling_logp_difference/max": 2.2510461807250977, "sampling/sampling_logp_difference/mean": 0.0785253494977951, "step": 1574 }, { "clip_ratio/high_max": 0.02857396099716425, "clip_ratio/high_mean": 0.02857396099716425, "clip_ratio/low_mean": 0.016170635353773832, "clip_ratio/low_min": 0.016170635353773832, "clip_ratio/region_mean": 0.04474459635093808, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.5, "completions/mean_terminated_length": 61.5, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.3381665423512459, "epoch": 0.06326063381130256, "frac_reward_zero_std": 0.0, "grad_norm": 6.546207427978516, "learning_rate": 5.230303030303031e-06, "loss": -0.0016, "num_tokens": 3536318.0, "reward": 0.9349583387374878, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9349583387374878, "reward_meter_std": 0.10497313737869263, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.10497313737869263, "reward_total_composite_mean": 0.9349583387374878, "reward_total_composite_std": 0.10497313737869263, "reward_total_mean": 0.9349583387374878, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9349583387374878, "rewards/meter/std": 0.10497313737869263, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9349583387374878, "rewards/total_composite/std": 0.10497313737869263, "sampling/importance_sampling_ratio/max": 1.711548924446106, "sampling/importance_sampling_ratio/mean": 1.0037705898284912, "sampling/importance_sampling_ratio/min": 0.2560369670391083, "sampling/sampling_logp_difference/max": 1.3624334335327148, "sampling/sampling_logp_difference/mean": 0.04767481982707977, "step": 1575 }, { "clip_ratio/high_max": 0.03661064524203539, "clip_ratio/high_mean": 0.03661064524203539, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/region_mean": 0.03856377024203539, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 61.75, "completions/mean_terminated_length": 61.75, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.42180035449564457, "epoch": 0.06330079929308752, "frac_reward_zero_std": 0.0, "grad_norm": 3.9961159229278564, "learning_rate": 5.2272727272727274e-06, "loss": 0.0111, "num_tokens": 3538180.0, "reward": 0.9147671461105347, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9147671461105347, "reward_meter_std": 0.21282289922237396, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.21282285451889038, "reward_total_composite_mean": 0.9147671461105347, "reward_total_composite_std": 0.21282289922237396, "reward_total_mean": 0.9147671461105347, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9147671461105347, "rewards/meter/std": 0.21282289922237396, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9147671461105347, "rewards/total_composite/std": 0.21282289922237396, "sampling/importance_sampling_ratio/max": 1.9876987934112549, "sampling/importance_sampling_ratio/mean": 1.0070737600326538, "sampling/importance_sampling_ratio/min": 0.20092912018299103, "sampling/sampling_logp_difference/max": 1.6048030853271484, "sampling/sampling_logp_difference/mean": 0.05491922050714493, "step": 1576 }, { "clip_ratio/high_max": 0.005319148767739534, "clip_ratio/high_mean": 0.005319148767739534, "clip_ratio/low_mean": 0.02302604599390179, "clip_ratio/low_min": 0.02302604599390179, "clip_ratio/region_mean": 0.028345194761641324, "completions/clipped_ratio": 0.0, "completions/max_length": 97.0, "completions/max_terminated_length": 97.0, "completions/mean_length": 92.0, "completions/mean_terminated_length": 92.0, "completions/min_length": 89.0, "completions/min_terminated_length": 89.0, "entropy": 0.24610379338264465, "epoch": 0.06334096477487247, "frac_reward_zero_std": 0.0, "grad_norm": 4.388180732727051, "learning_rate": 5.224242424242425e-06, "loss": 0.0008, "num_tokens": 3540188.0, "reward": 0.8181818723678589, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9917011260986328, "reward_meter_std": 0.006606912240386009, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07076937705278397, "reward_total_composite_mean": 0.8181818723678589, "reward_total_composite_std": 0.07076939195394516, "reward_total_mean": 0.8181818723678589, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9917011260986328, "rewards/meter/std": 0.006606912240386009, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8181818723678589, "rewards/total_composite/std": 0.07076939195394516, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0047838687896729, "sampling/importance_sampling_ratio/min": 0.3042147159576416, "sampling/sampling_logp_difference/max": 1.1900215148925781, "sampling/sampling_logp_difference/mean": 0.037422213703393936, "step": 1577 }, { "clip_ratio/high_max": 0.02987705054692924, "clip_ratio/high_mean": 0.02987705054692924, "clip_ratio/low_mean": 0.012475505471229553, "clip_ratio/low_min": 0.012475505471229553, "clip_ratio/region_mean": 0.04235255601815879, "completions/clipped_ratio": 0.0, "completions/max_length": 218.0, "completions/max_terminated_length": 218.0, "completions/mean_length": 191.0, "completions/mean_terminated_length": 191.0, "completions/min_length": 175.0, "completions/min_terminated_length": 175.0, "entropy": 0.4086650311946869, "epoch": 0.06338113025665743, "frac_reward_zero_std": 0.0, "grad_norm": 4.709424018859863, "learning_rate": 5.221212121212121e-06, "loss": 0.0638, "num_tokens": 3543292.0, "reward": 0.8337088823318481, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.1035098284482956, "reward_meter_mean": 0.9755741953849792, "reward_meter_std": 0.027564238756895065, "reward_repeat_penalty_mean": 0.9154040217399597, "reward_repeat_penalty_std": 0.10342530906200409, "reward_std": 0.17133086919784546, "reward_total_composite_mean": 0.8337088823318481, "reward_total_composite_std": 0.17133085429668427, "reward_total_mean": 0.8337088823318481, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.1035098284482956, "rewards/meter/mean": 0.9755741953849792, "rewards/meter/std": 0.027564238756895065, "rewards/repeat_penalty/mean": 0.9154040217399597, "rewards/repeat_penalty/std": 0.10342530906200409, "rewards/total_composite/mean": 0.8337088823318481, "rewards/total_composite/std": 0.17133085429668427, "sampling/importance_sampling_ratio/max": 1.927268385887146, "sampling/importance_sampling_ratio/mean": 1.0094478130340576, "sampling/importance_sampling_ratio/min": 0.06850247830152512, "sampling/sampling_logp_difference/max": 2.6808853149414062, "sampling/sampling_logp_difference/mean": 0.05211658403277397, "step": 1578 }, { "clip_ratio/high_max": 0.0012376237427815795, "clip_ratio/high_mean": 0.0012376237427815795, "clip_ratio/low_mean": 0.003676470718346536, "clip_ratio/low_min": 0.003676470718346536, "clip_ratio/region_mean": 0.004914094461128116, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 100.625, "completions/mean_terminated_length": 100.625, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.042127607855945826, "epoch": 0.06342129573844238, "frac_reward_zero_std": 0.0, "grad_norm": 2.8309895992279053, "learning_rate": 5.218181818181819e-06, "loss": 0.008, "num_tokens": 3545353.0, "reward": 0.9478673934936523, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977493286132812, "reward_meter_std": 0.00013334653340280056, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09243172407150269, "reward_total_composite_mean": 0.9478673934936523, "reward_total_composite_std": 0.09243174642324448, "reward_total_mean": 0.9478673934936523, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977493286132812, "rewards/meter/std": 0.00013334653340280056, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9478673934936523, "rewards/total_composite/std": 0.09243174642324448, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0029453039169312, "sampling/importance_sampling_ratio/min": 0.5253335237503052, "sampling/sampling_logp_difference/max": 0.8175020217895508, "sampling/sampling_logp_difference/mean": 0.007003183010965586, "step": 1579 }, { "clip_ratio/high_max": 0.03686255170032382, "clip_ratio/high_mean": 0.03686255170032382, "clip_ratio/low_mean": 0.028440887574106455, "clip_ratio/low_min": 0.028440887574106455, "clip_ratio/region_mean": 0.06530343927443027, "completions/clipped_ratio": 0.0, "completions/max_length": 243.0, "completions/max_terminated_length": 243.0, "completions/mean_length": 233.25, "completions/mean_terminated_length": 233.25, "completions/min_length": 217.0, "completions/min_terminated_length": 217.0, "entropy": 1.0062834583222866, "epoch": 0.06346146122022733, "frac_reward_zero_std": 0.0, "grad_norm": 3.488912343978882, "learning_rate": 5.215151515151516e-06, "loss": 0.017, "num_tokens": 3548859.0, "reward": 0.6222666501998901, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6562398672103882, "reward_meter_std": 0.2839363217353821, "reward_repeat_penalty_mean": 0.9615384340286255, "reward_repeat_penalty_std": 0.082234226167202, "reward_std": 0.2602792978286743, "reward_total_composite_mean": 0.6222666501998901, "reward_total_composite_std": 0.2602792978286743, "reward_total_mean": 0.6222666501998901, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6562398672103882, "rewards/meter/std": 0.2839363217353821, "rewards/repeat_penalty/mean": 0.9615384340286255, "rewards/repeat_penalty/std": 0.082234226167202, "rewards/total_composite/mean": 0.6222666501998901, "rewards/total_composite/std": 0.2602792978286743, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0219330787658691, "sampling/importance_sampling_ratio/min": 0.22948291897773743, "sampling/sampling_logp_difference/max": 1.4719266891479492, "sampling/sampling_logp_difference/mean": 0.09925277531147003, "step": 1580 }, { "clip_ratio/high_max": 0.0039564905455335975, "clip_ratio/high_mean": 0.0039564905455335975, "clip_ratio/low_mean": 0.00662941113114357, "clip_ratio/low_min": 0.00662941113114357, "clip_ratio/region_mean": 0.010585901676677167, "completions/clipped_ratio": 0.0, "completions/max_length": 255.0, "completions/max_terminated_length": 255.0, "completions/mean_length": 250.25, "completions/mean_terminated_length": 250.25, "completions/min_length": 240.0, "completions/min_terminated_length": 240.0, "entropy": 0.12574958708137274, "epoch": 0.06350162670201229, "frac_reward_zero_std": 0.0, "grad_norm": 1.136604905128479, "learning_rate": 5.212121212121213e-06, "loss": -0.0056, "num_tokens": 3552445.0, "reward": 0.7205430865287781, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991568922996521, "reward_meter_std": 0.00013457036402542144, "reward_repeat_penalty_mean": 0.7211538553237915, "reward_repeat_penalty_std": 0.039811473339796066, "reward_std": 0.03971904516220093, "reward_total_composite_mean": 0.7205430865287781, "reward_total_composite_std": 0.039719030261039734, "reward_total_mean": 0.7205430865287781, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991568922996521, "rewards/meter/std": 0.00013457036402542144, "rewards/repeat_penalty/mean": 0.7211538553237915, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.7205430865287781, "rewards/total_composite/std": 0.039719030261039734, "sampling/importance_sampling_ratio/max": 1.794761300086975, "sampling/importance_sampling_ratio/mean": 1.0050798654556274, "sampling/importance_sampling_ratio/min": 0.14707794785499573, "sampling/sampling_logp_difference/max": 1.916792631149292, "sampling/sampling_logp_difference/mean": 0.017896855250000954, "step": 1581 }, { "clip_ratio/high_max": 0.034577128011733294, "clip_ratio/high_mean": 0.034577128011733294, "clip_ratio/low_mean": 0.010625278111547232, "clip_ratio/low_min": 0.010625278111547232, "clip_ratio/region_mean": 0.045202406123280525, "completions/clipped_ratio": 0.0, "completions/max_length": 112.0, "completions/max_terminated_length": 112.0, "completions/mean_length": 107.5, "completions/mean_terminated_length": 107.5, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.4601043425500393, "epoch": 0.06354179218379724, "frac_reward_zero_std": 0.0, "grad_norm": 4.802741527557373, "learning_rate": 5.209090909090909e-06, "loss": 0.0002, "num_tokens": 3554625.0, "reward": 0.989787757396698, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.989787757396698, "reward_meter_std": 0.00834440253674984, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008344405330717564, "reward_total_composite_mean": 0.989787757396698, "reward_total_composite_std": 0.00834440253674984, "reward_total_mean": 0.989787757396698, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.989787757396698, "rewards/meter/std": 0.00834440253674984, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.989787757396698, "rewards/total_composite/std": 0.00834440253674984, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.015655517578125, "sampling/importance_sampling_ratio/min": 0.24133498966693878, "sampling/sampling_logp_difference/max": 1.5647554397583008, "sampling/sampling_logp_difference/mean": 0.05708528682589531, "step": 1582 }, { "clip_ratio/high_max": 0.03560801479034126, "clip_ratio/high_mean": 0.03560801479034126, "clip_ratio/low_mean": 0.010714286006987095, "clip_ratio/low_min": 0.010714286006987095, "clip_ratio/region_mean": 0.04632230079732835, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.25, "completions/mean_terminated_length": 67.25, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.5009849853813648, "epoch": 0.0635819576655822, "frac_reward_zero_std": 0.0, "grad_norm": 13.880090713500977, "learning_rate": 5.2060606060606065e-06, "loss": 0.0415, "num_tokens": 3556523.0, "reward": 0.7919849157333374, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7919849157333374, "reward_meter_std": 0.3320651650428772, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3320651352405548, "reward_total_composite_mean": 0.7919849157333374, "reward_total_composite_std": 0.3320651650428772, "reward_total_mean": 0.7919849157333374, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7919849157333374, "rewards/meter/std": 0.3320651650428772, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7919849157333374, "rewards/total_composite/std": 0.3320651650428772, "sampling/importance_sampling_ratio/max": 1.9709203243255615, "sampling/importance_sampling_ratio/mean": 1.0017415285110474, "sampling/importance_sampling_ratio/min": 0.22323958575725555, "sampling/sampling_logp_difference/max": 1.4995098114013672, "sampling/sampling_logp_difference/mean": 0.060586389154195786, "step": 1583 }, { "clip_ratio/high_max": 0.010113696102052927, "clip_ratio/high_mean": 0.010113696102052927, "clip_ratio/low_mean": 0.006222632946446538, "clip_ratio/low_min": 0.006222632946446538, "clip_ratio/region_mean": 0.016336329048499465, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 60.75, "completions/mean_terminated_length": 60.75, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.22835318371653557, "epoch": 0.06362212314736715, "frac_reward_zero_std": 0.0, "grad_norm": 4.862657070159912, "learning_rate": 5.203030303030303e-06, "loss": 0.0011, "num_tokens": 3558329.0, "reward": 0.9940457940101624, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9940457940101624, "reward_meter_std": 0.0014534186339005828, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0014534194488078356, "reward_total_composite_mean": 0.9940457940101624, "reward_total_composite_std": 0.0014534186339005828, "reward_total_mean": 0.9940457940101624, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9940457940101624, "rewards/meter/std": 0.0014534186339005828, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9940457940101624, "rewards/total_composite/std": 0.0014534186339005828, "sampling/importance_sampling_ratio/max": 1.5038725137710571, "sampling/importance_sampling_ratio/mean": 1.0062899589538574, "sampling/importance_sampling_ratio/min": 0.28087490797042847, "sampling/sampling_logp_difference/max": 1.269845962524414, "sampling/sampling_logp_difference/mean": 0.02881665527820587, "step": 1584 }, { "clip_ratio/high_max": 0.007812500116415322, "clip_ratio/high_mean": 0.007812500116415322, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007812500116415322, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 144.0, "completions/mean_terminated_length": 144.0, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.0726482942700386, "epoch": 0.0636622886291521, "frac_reward_zero_std": 0.0, "grad_norm": 2.1006689071655273, "learning_rate": 5.2e-06, "loss": 0.0019, "num_tokens": 3560777.0, "reward": 0.8384406566619873, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989944696426392, "reward_meter_std": 1.850287117122207e-05, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.09155284613370895, "reward_std": 0.09144823998212814, "reward_total_composite_mean": 0.8384406566619873, "reward_total_composite_std": 0.09144823998212814, "reward_total_mean": 0.8384406566619873, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989944696426392, "rewards/meter/std": 1.850287117122207e-05, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.8384406566619873, "rewards/total_composite/std": 0.09144823998212814, "sampling/importance_sampling_ratio/max": 1.262144923210144, "sampling/importance_sampling_ratio/mean": 1.0028645992279053, "sampling/importance_sampling_ratio/min": 0.4136001169681549, "sampling/sampling_logp_difference/max": 0.8828556537628174, "sampling/sampling_logp_difference/mean": 0.010658186860382557, "step": 1585 }, { "clip_ratio/high_max": 0.03553921659477055, "clip_ratio/high_mean": 0.03553921659477055, "clip_ratio/low_mean": 0.017196210101246834, "clip_ratio/low_min": 0.017196210101246834, "clip_ratio/region_mean": 0.052735426696017385, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 66.625, "completions/mean_terminated_length": 66.625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.6144770756363869, "epoch": 0.06370245411093706, "frac_reward_zero_std": 0.0, "grad_norm": 6.0716328620910645, "learning_rate": 5.196969696969697e-06, "loss": 0.0059, "num_tokens": 3562502.0, "reward": 0.8443154096603394, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8443154096603394, "reward_meter_std": 0.19778741896152496, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.19778741896152496, "reward_total_composite_mean": 0.8443154096603394, "reward_total_composite_std": 0.19778741896152496, "reward_total_mean": 0.8443154096603394, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8443154096603394, "rewards/meter/std": 0.19778741896152496, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8443154096603394, "rewards/total_composite/std": 0.19778741896152496, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0118776559829712, "sampling/importance_sampling_ratio/min": 0.3133116662502289, "sampling/sampling_logp_difference/max": 1.1605567932128906, "sampling/sampling_logp_difference/mean": 0.0697539895772934, "step": 1586 }, { "clip_ratio/high_max": 0.03026041854172945, "clip_ratio/high_mean": 0.03026041854172945, "clip_ratio/low_mean": 0.01317360415123403, "clip_ratio/low_min": 0.01317360415123403, "clip_ratio/region_mean": 0.04343402269296348, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.5464729219675064, "epoch": 0.06374261959272201, "frac_reward_zero_std": 0.0, "grad_norm": 5.30000114440918, "learning_rate": 5.193939393939395e-06, "loss": -0.0017, "num_tokens": 3564369.0, "reward": 0.9182997941970825, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9182997941970825, "reward_meter_std": 0.08735460042953491, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.08735460788011551, "reward_total_composite_mean": 0.9182997941970825, "reward_total_composite_std": 0.08735460042953491, "reward_total_mean": 0.9182997941970825, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9182997941970825, "rewards/meter/std": 0.08735460042953491, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9182997941970825, "rewards/total_composite/std": 0.08735460042953491, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0096309185028076, "sampling/importance_sampling_ratio/min": 0.16284604370594025, "sampling/sampling_logp_difference/max": 1.8149499893188477, "sampling/sampling_logp_difference/mean": 0.06299924105405807, "step": 1587 }, { "clip_ratio/high_max": 0.025014446233399212, "clip_ratio/high_mean": 0.025014446233399212, "clip_ratio/low_mean": 0.007130191195756197, "clip_ratio/low_min": 0.007130191195756197, "clip_ratio/region_mean": 0.03214463742915541, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 69.875, "completions/mean_terminated_length": 69.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.4779812954366207, "epoch": 0.06378278507450696, "frac_reward_zero_std": 0.0, "grad_norm": 5.146764755249023, "learning_rate": 5.190909090909091e-06, "loss": 0.0024, "num_tokens": 3566296.0, "reward": 0.6954999566078186, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6954999566078186, "reward_meter_std": 0.4150845408439636, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.41508448123931885, "reward_total_composite_mean": 0.6954999566078186, "reward_total_composite_std": 0.4150845408439636, "reward_total_mean": 0.6954999566078186, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6954999566078186, "rewards/meter/std": 0.4150845408439636, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6954999566078186, "rewards/total_composite/std": 0.4150845408439636, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.014699101448059, "sampling/importance_sampling_ratio/min": 0.2578465938568115, "sampling/sampling_logp_difference/max": 1.3553905487060547, "sampling/sampling_logp_difference/mean": 0.0591333732008934, "step": 1588 }, { "clip_ratio/high_max": 0.04341738054063171, "clip_ratio/high_mean": 0.04341738054063171, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.04341738054063171, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.4802176021039486, "epoch": 0.06382295055629192, "frac_reward_zero_std": 0.0, "grad_norm": 5.699836730957031, "learning_rate": 5.187878787878788e-06, "loss": -0.0052, "num_tokens": 3568219.0, "reward": 0.8743665218353271, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8743665218353271, "reward_meter_std": 0.338323175907135, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3383232057094574, "reward_total_composite_mean": 0.8743665218353271, "reward_total_composite_std": 0.338323175907135, "reward_total_mean": 0.8743665218353271, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8743665218353271, "rewards/meter/std": 0.338323175907135, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8743665218353271, "rewards/total_composite/std": 0.338323175907135, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0109214782714844, "sampling/importance_sampling_ratio/min": 0.275927871465683, "sampling/sampling_logp_difference/max": 1.2876157760620117, "sampling/sampling_logp_difference/mean": 0.060609057545661926, "step": 1589 }, { "clip_ratio/high_max": 0.004474195709917694, "clip_ratio/high_mean": 0.004474195709917694, "clip_ratio/low_mean": 0.007355856709182262, "clip_ratio/low_min": 0.007355856709182262, "clip_ratio/region_mean": 0.011830052419099957, "completions/clipped_ratio": 0.0, "completions/max_length": 287.0, "completions/max_terminated_length": 287.0, "completions/mean_length": 271.75, "completions/mean_terminated_length": 271.75, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.08188144443556666, "epoch": 0.06386311603807687, "frac_reward_zero_std": 0.0, "grad_norm": 1.8666749000549316, "learning_rate": 5.184848484848485e-06, "loss": -0.0196, "num_tokens": 3571993.0, "reward": 0.5278421640396118, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8611111640930176, "reward_count_adherence_std": 0.05143444612622261, "reward_meter_mean": 0.9924838542938232, "reward_meter_std": 0.004395514260977507, "reward_repeat_penalty_mean": 0.6185267567634583, "reward_repeat_penalty_std": 0.05406184867024422, "reward_std": 0.048469796776771545, "reward_total_composite_mean": 0.5278421640396118, "reward_total_composite_std": 0.04846978932619095, "reward_total_mean": 0.5278421640396118, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8611111640930176, "rewards/count_adherence/std": 0.05143444612622261, "rewards/meter/mean": 0.9924838542938232, "rewards/meter/std": 0.004395514260977507, "rewards/repeat_penalty/mean": 0.6185267567634583, "rewards/repeat_penalty/std": 0.05406184867024422, "rewards/total_composite/mean": 0.5278421640396118, "rewards/total_composite/std": 0.04846978932619095, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0014569759368896, "sampling/importance_sampling_ratio/min": 0.004255921114236116, "sampling/sampling_logp_difference/max": 5.459444046020508, "sampling/sampling_logp_difference/mean": 0.018268149346113205, "step": 1590 }, { "clip_ratio/high_max": 0.03321157908067107, "clip_ratio/high_mean": 0.03321157908067107, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.03321157908067107, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.75, "completions/mean_terminated_length": 67.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.48275283351540565, "epoch": 0.06390328151986183, "frac_reward_zero_std": 0.0, "grad_norm": 6.771106719970703, "learning_rate": 5.181818181818182e-06, "loss": 0.0162, "num_tokens": 3573831.0, "reward": 0.89401775598526, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.89401775598526, "reward_meter_std": 0.16673576831817627, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.16673575341701508, "reward_total_composite_mean": 0.89401775598526, "reward_total_composite_std": 0.16673576831817627, "reward_total_mean": 0.89401775598526, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.89401775598526, "rewards/meter/std": 0.16673576831817627, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.89401775598526, "rewards/total_composite/std": 0.16673576831817627, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0108517408370972, "sampling/importance_sampling_ratio/min": 0.40265539288520813, "sampling/sampling_logp_difference/max": 1.1782063245773315, "sampling/sampling_logp_difference/mean": 0.05256752669811249, "step": 1591 }, { "clip_ratio/high_max": 0.006465517217293382, "clip_ratio/high_mean": 0.006465517217293382, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/region_mean": 0.008584161289036274, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 58.125, "completions/mean_terminated_length": 58.125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.020813994109630585, "epoch": 0.06394344700164678, "frac_reward_zero_std": 0.0, "grad_norm": 1.81292724609375, "learning_rate": 5.1787878787878784e-06, "loss": 0.0026, "num_tokens": 3575592.0, "reward": 0.9933634400367737, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9933634400367737, "reward_meter_std": 0.00030929798958823085, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00030929522472433746, "reward_total_composite_mean": 0.9933634400367737, "reward_total_composite_std": 0.00030929798958823085, "reward_total_mean": 0.9933634400367737, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9933634400367737, "rewards/meter/std": 0.00030929798958823085, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9933634400367737, "rewards/total_composite/std": 0.00030929798958823085, "sampling/importance_sampling_ratio/max": 1.1543108224868774, "sampling/importance_sampling_ratio/mean": 0.9998379349708557, "sampling/importance_sampling_ratio/min": 0.4771060645580292, "sampling/sampling_logp_difference/max": 0.7400164604187012, "sampling/sampling_logp_difference/mean": 0.004956495948135853, "step": 1592 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.012177230441011488, "clip_ratio/low_min": 0.012177230441011488, "clip_ratio/region_mean": 0.01404290203936398, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 68.75, "completions/mean_terminated_length": 68.75, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.12029994325712323, "epoch": 0.06398361248343173, "frac_reward_zero_std": 0.0, "grad_norm": 10.237826347351074, "learning_rate": 5.1757575757575765e-06, "loss": 0.035, "num_tokens": 3577358.0, "reward": 0.7235053181648254, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7235053181648254, "reward_meter_std": 0.3811579644680023, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3811579942703247, "reward_total_composite_mean": 0.7235053181648254, "reward_total_composite_std": 0.3811579644680023, "reward_total_mean": 0.7235053181648254, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7235053181648254, "rewards/meter/std": 0.3811579644680023, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7235053181648254, "rewards/total_composite/std": 0.3811579644680023, "sampling/importance_sampling_ratio/max": 1.7127673625946045, "sampling/importance_sampling_ratio/mean": 0.9999585151672363, "sampling/importance_sampling_ratio/min": 0.17401038110256195, "sampling/sampling_logp_difference/max": 1.7486402988433838, "sampling/sampling_logp_difference/mean": 0.027932489290833473, "step": 1593 }, { "clip_ratio/high_max": 0.007002223108429462, "clip_ratio/high_mean": 0.007002223108429462, "clip_ratio/low_mean": 0.005208890186622739, "clip_ratio/low_min": 0.005208890186622739, "clip_ratio/region_mean": 0.0122111132950522, "completions/clipped_ratio": 0.0, "completions/max_length": 200.0, "completions/max_terminated_length": 200.0, "completions/mean_length": 195.375, "completions/mean_terminated_length": 195.375, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.06844896962866187, "epoch": 0.06402377796521669, "frac_reward_zero_std": 0.0, "grad_norm": 1.7681658267974854, "learning_rate": 5.172727272727273e-06, "loss": 0.0095, "num_tokens": 3580425.0, "reward": 0.7389621734619141, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953435659408569, "reward_meter_std": 0.0010429132962599397, "reward_repeat_penalty_mean": 0.7424242496490479, "reward_repeat_penalty_std": 0.04169124737381935, "reward_std": 0.041448723524808884, "reward_total_composite_mean": 0.7389621734619141, "reward_total_composite_std": 0.04144872725009918, "reward_total_mean": 0.7389621734619141, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953435659408569, "rewards/meter/std": 0.0010429132962599397, "rewards/repeat_penalty/mean": 0.7424242496490479, "rewards/repeat_penalty/std": 0.04169124737381935, "rewards/total_composite/mean": 0.7389621734619141, "rewards/total_composite/std": 0.04144872725009918, "sampling/importance_sampling_ratio/max": 1.846528172492981, "sampling/importance_sampling_ratio/mean": 0.9993682503700256, "sampling/importance_sampling_ratio/min": 0.10044170916080475, "sampling/sampling_logp_difference/max": 2.298177719116211, "sampling/sampling_logp_difference/mean": 0.016124166548252106, "step": 1594 }, { "clip_ratio/high_max": 0.030346576124429703, "clip_ratio/high_mean": 0.030346576124429703, "clip_ratio/low_mean": 0.018327449448406696, "clip_ratio/low_min": 0.018327449448406696, "clip_ratio/region_mean": 0.0486740255728364, "completions/clipped_ratio": 0.0, "completions/max_length": 194.0, "completions/max_terminated_length": 194.0, "completions/mean_length": 161.625, "completions/mean_terminated_length": 161.625, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.49944454059004784, "epoch": 0.06406394344700164, "frac_reward_zero_std": 0.0, "grad_norm": 4.416578769683838, "learning_rate": 5.16969696969697e-06, "loss": 0.0842, "num_tokens": 3583190.0, "reward": 0.8883745670318604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_meter_mean": 0.9966393113136292, "reward_meter_std": 0.0009264187538065016, "reward_repeat_penalty_mean": 0.9503968358039856, "reward_repeat_penalty_std": 0.06915634870529175, "reward_std": 0.13054178655147552, "reward_total_composite_mean": 0.8883745670318604, "reward_total_composite_std": 0.13054178655147552, "reward_total_mean": 0.8883745670318604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/meter/mean": 0.9966393113136292, "rewards/meter/std": 0.0009264187538065016, "rewards/repeat_penalty/mean": 0.9503968358039856, "rewards/repeat_penalty/std": 0.06915634870529175, "rewards/total_composite/mean": 0.8883745670318604, "rewards/total_composite/std": 0.13054178655147552, "sampling/importance_sampling_ratio/max": 1.7064915895462036, "sampling/importance_sampling_ratio/mean": 1.0092458724975586, "sampling/importance_sampling_ratio/min": 0.06590858101844788, "sampling/sampling_logp_difference/max": 2.719486713409424, "sampling/sampling_logp_difference/mean": 0.0641026571393013, "step": 1595 }, { "clip_ratio/high_max": 0.02265306143090129, "clip_ratio/high_mean": 0.02265306143090129, "clip_ratio/low_mean": 0.010051020188257098, "clip_ratio/low_min": 0.010051020188257098, "clip_ratio/region_mean": 0.03270408161915839, "completions/clipped_ratio": 0.0, "completions/max_length": 50.0, "completions/max_terminated_length": 50.0, "completions/mean_length": 49.75, "completions/mean_terminated_length": 49.75, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.11144222784787416, "epoch": 0.0641041089287866, "frac_reward_zero_std": 0.0, "grad_norm": 6.151648998260498, "learning_rate": 5.1666666666666675e-06, "loss": 0.0073, "num_tokens": 3584948.0, "reward": 0.9367859363555908, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9367859363555908, "reward_meter_std": 0.005013443063944578, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005013437941670418, "reward_total_composite_mean": 0.9367859363555908, "reward_total_composite_std": 0.005013443063944578, "reward_total_mean": 0.9367859363555908, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9367859363555908, "rewards/meter/std": 0.005013443063944578, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9367859363555908, "rewards/total_composite/std": 0.005013443063944578, "sampling/importance_sampling_ratio/max": 1.7191351652145386, "sampling/importance_sampling_ratio/mean": 1.0016454458236694, "sampling/importance_sampling_ratio/min": 0.2862391471862793, "sampling/sampling_logp_difference/max": 1.2509276866912842, "sampling/sampling_logp_difference/mean": 0.024813249707221985, "step": 1596 }, { "clip_ratio/high_max": 0.02771200449205935, "clip_ratio/high_mean": 0.02771200449205935, "clip_ratio/low_mean": 0.011171216145157814, "clip_ratio/low_min": 0.011171216145157814, "clip_ratio/region_mean": 0.038883220637217164, "completions/clipped_ratio": 0.0, "completions/max_length": 117.0, "completions/max_terminated_length": 117.0, "completions/mean_length": 111.875, "completions/mean_terminated_length": 111.875, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.40937453880906105, "epoch": 0.06414427441057155, "frac_reward_zero_std": 0.0, "grad_norm": 4.290965557098389, "learning_rate": 5.163636363636364e-06, "loss": 0.0102, "num_tokens": 3587243.0, "reward": 0.9206209182739258, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952096939086914, "reward_meter_std": 0.004024881403893232, "reward_repeat_penalty_mean": 0.925000011920929, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10357693582773209, "reward_total_composite_mean": 0.9206209182739258, "reward_total_composite_std": 0.1035769134759903, "reward_total_mean": 0.9206209182739258, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952096939086914, "rewards/meter/std": 0.004024881403893232, "rewards/repeat_penalty/mean": 0.925000011920929, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.9206209182739258, "rewards/total_composite/std": 0.1035769134759903, "sampling/importance_sampling_ratio/max": 1.8452407121658325, "sampling/importance_sampling_ratio/mean": 1.011698842048645, "sampling/importance_sampling_ratio/min": 0.14579784870147705, "sampling/sampling_logp_difference/max": 1.9255342483520508, "sampling/sampling_logp_difference/mean": 0.052159715443849564, "step": 1597 }, { "clip_ratio/high_max": 0.01751898298971355, "clip_ratio/high_mean": 0.01751898298971355, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/region_mean": 0.01947210798971355, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.0873101712204516, "epoch": 0.0641844398923565, "frac_reward_zero_std": 0.0, "grad_norm": 2.746960401535034, "learning_rate": 5.160606060606061e-06, "loss": 0.003, "num_tokens": 3589155.0, "reward": 0.9953999519348145, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953999519348145, "reward_meter_std": 0.0069043696857988834, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006904360372573137, "reward_total_composite_mean": 0.9953999519348145, "reward_total_composite_std": 0.0069043696857988834, "reward_total_mean": 0.9953999519348145, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953999519348145, "rewards/meter/std": 0.0069043696857988834, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953999519348145, "rewards/total_composite/std": 0.0069043696857988834, "sampling/importance_sampling_ratio/max": 1.8938157558441162, "sampling/importance_sampling_ratio/mean": 0.9994773268699646, "sampling/importance_sampling_ratio/min": 0.42812401056289673, "sampling/sampling_logp_difference/max": 0.8483424186706543, "sampling/sampling_logp_difference/mean": 0.021562010049819946, "step": 1598 }, { "clip_ratio/high_max": 0.017178362933918834, "clip_ratio/high_mean": 0.017178362933918834, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/region_mean": 0.023935119854286313, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 37.0, "completions/mean_terminated_length": 37.0, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.307000532746315, "epoch": 0.06422460537414146, "frac_reward_zero_std": 0.0, "grad_norm": 6.9674506187438965, "learning_rate": 5.1575757575757575e-06, "loss": 0.0135, "num_tokens": 3590515.0, "reward": 0.9930006265640259, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9930006265640259, "reward_meter_std": 0.00047757019638083875, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004775776178576052, "reward_total_composite_mean": 0.9930006265640259, "reward_total_composite_std": 0.00047757019638083875, "reward_total_mean": 0.9930006265640259, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9930006265640259, "rewards/meter/std": 0.00047757019638083875, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9930006265640259, "rewards/total_composite/std": 0.00047757019638083875, "sampling/importance_sampling_ratio/max": 1.5318171977996826, "sampling/importance_sampling_ratio/mean": 1.0091636180877686, "sampling/importance_sampling_ratio/min": 0.6027238368988037, "sampling/sampling_logp_difference/max": 0.5062961578369141, "sampling/sampling_logp_difference/mean": 0.030742382630705833, "step": 1599 }, { "clip_ratio/high_max": 0.01740056835114956, "clip_ratio/high_mean": 0.01740056835114956, "clip_ratio/low_mean": 0.007490954361855984, "clip_ratio/low_min": 0.007490954361855984, "clip_ratio/region_mean": 0.024891522713005543, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 64.875, "completions/mean_terminated_length": 64.875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.24194820784032345, "epoch": 0.06426477085592641, "frac_reward_zero_std": 0.0, "grad_norm": 6.287996292114258, "learning_rate": 5.154545454545456e-06, "loss": 0.0215, "num_tokens": 3592322.0, "reward": 0.9355560541152954, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9355560541152954, "reward_meter_std": 0.06913071125745773, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06913069635629654, "reward_total_composite_mean": 0.9355560541152954, "reward_total_composite_std": 0.06913071125745773, "reward_total_mean": 0.9355560541152954, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9355560541152954, "rewards/meter/std": 0.06913071125745773, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9355560541152954, "rewards/total_composite/std": 0.06913071125745773, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003296136856079, "sampling/importance_sampling_ratio/min": 0.3875372111797333, "sampling/sampling_logp_difference/max": 0.9479434490203857, "sampling/sampling_logp_difference/mean": 0.03889580816030502, "step": 1600 }, { "epoch": 0.06426477085592641, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 347.3076923076923, "eval_completions/max_terminated_length": 347.3076923076923, "eval_completions/mean_length": 192.58653846153845, "eval_completions/mean_terminated_length": 192.58653846153845, "eval_completions/min_length": 61.76923076923077, "eval_completions/min_terminated_length": 61.76923076923077, "eval_entropy": 0.19453936872574, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 3592322.0, "eval_reward": 0.5311165887575883, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9018935148532574, "eval_reward_count_adherence_std": 0.12116303610113952, "eval_reward_meter_mean": 0.7114841112723718, "eval_reward_meter_std": 0.38725116863273656, "eval_reward_repeat_penalty_mean": 0.8302872364337628, "eval_reward_repeat_penalty_std": 0.14936749923687714, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5311165887575883, "eval_reward_total_composite_std": 0.34226179122924805, "eval_reward_total_mean": 0.5311165887575883, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9018935148532574, "eval_rewards/count_adherence/std": 0.12116303610113952, "eval_rewards/meter/mean": 0.7114841112723718, "eval_rewards/meter/std": 0.38725116863273656, "eval_rewards/repeat_penalty/mean": 0.8302872364337628, "eval_rewards/repeat_penalty/std": 0.14936749923687714, "eval_rewards/total_composite/mean": 0.5311165887575883, "eval_rewards/total_composite/std": 0.34226179122924805, "eval_runtime": 66.2489, "eval_samples_per_second": 1.57, "eval_sampling/importance_sampling_ratio/max": 1.4436852565178504, "eval_sampling/importance_sampling_ratio/mean": 1.004677598293011, "eval_sampling/importance_sampling_ratio/min": 0.35776989047343916, "eval_sampling/sampling_logp_difference/max": 1.060350748208853, "eval_sampling/sampling_logp_difference/mean": 0.0201402991436995, "eval_steps_per_second": 0.196, "step": 1600 }, { "clip_ratio/high_max": 0.03826867160387337, "clip_ratio/high_mean": 0.03826867160387337, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/region_mean": 0.04169332911260426, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 77.25, "completions/mean_terminated_length": 77.25, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.4462718777358532, "epoch": 0.06430493633771137, "frac_reward_zero_std": 0.0, "grad_norm": 5.907294273376465, "learning_rate": 5.151515151515152e-06, "loss": -0.0132, "num_tokens": 3594188.0, "reward": 0.9925735592842102, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9925735592842102, "reward_meter_std": 0.013519754633307457, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.013519756495952606, "reward_total_composite_mean": 0.9925735592842102, "reward_total_composite_std": 0.013519754633307457, "reward_total_mean": 0.9925735592842102, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9925735592842102, "rewards/meter/std": 0.013519754633307457, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9925735592842102, "rewards/total_composite/std": 0.013519754633307457, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0113800764083862, "sampling/importance_sampling_ratio/min": 0.21933509409427643, "sampling/sampling_logp_difference/max": 3.263657808303833, "sampling/sampling_logp_difference/mean": 0.061646249145269394, "step": 1601 }, { "clip_ratio/high_max": 0.016364703187718987, "clip_ratio/high_mean": 0.016364703187718987, "clip_ratio/low_mean": 0.038138989359140396, "clip_ratio/low_min": 0.038138989359140396, "clip_ratio/region_mean": 0.054503692546859384, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 133.0, "completions/mean_terminated_length": 133.0, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.4912920892238617, "epoch": 0.06434510181949632, "frac_reward_zero_std": 0.0, "grad_norm": 3.9016387462615967, "learning_rate": 5.148484848484849e-06, "loss": -0.0111, "num_tokens": 3596572.0, "reward": 0.08772411197423935, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.10229554772377014, "reward_meter_std": 0.20073863863945007, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.17203769087791443, "reward_total_composite_mean": 0.08772411197423935, "reward_total_composite_std": 0.17203770577907562, "reward_total_mean": 0.08772411197423935, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.10229554772377014, "rewards/meter/std": 0.20073863863945007, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.08772411197423935, "rewards/total_composite/std": 0.17203770577907562, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0097557306289673, "sampling/importance_sampling_ratio/min": 0.27479445934295654, "sampling/sampling_logp_difference/max": 1.3837254047393799, "sampling/sampling_logp_difference/mean": 0.06598487496376038, "step": 1602 }, { "clip_ratio/high_max": 0.002314814832061529, "clip_ratio/high_mean": 0.002314814832061529, "clip_ratio/low_mean": 0.01308055012486875, "clip_ratio/low_min": 0.01308055012486875, "clip_ratio/region_mean": 0.01539536495693028, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 104.875, "completions/mean_terminated_length": 104.875, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.10506064165383577, "epoch": 0.06438526730128127, "frac_reward_zero_std": 0.0, "grad_norm": 2.740659713745117, "learning_rate": 5.145454545454546e-06, "loss": -0.0097, "num_tokens": 3598835.0, "reward": 0.8239848613739014, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987785816192627, "reward_meter_std": 0.00020610357751138508, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07050740718841553, "reward_total_composite_mean": 0.8239848613739014, "reward_total_composite_std": 0.07050739973783493, "reward_total_mean": 0.8239848613739014, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987785816192627, "rewards/meter/std": 0.00020610357751138508, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8239848613739014, "rewards/total_composite/std": 0.07050739973783493, "sampling/importance_sampling_ratio/max": 1.973273754119873, "sampling/importance_sampling_ratio/mean": 1.0108842849731445, "sampling/importance_sampling_ratio/min": 0.5668526887893677, "sampling/sampling_logp_difference/max": 0.6796939373016357, "sampling/sampling_logp_difference/mean": 0.019803691655397415, "step": 1603 }, { "clip_ratio/high_max": 0.006316610379144549, "clip_ratio/high_mean": 0.006316610379144549, "clip_ratio/low_mean": 0.00780038780067116, "clip_ratio/low_min": 0.00780038780067116, "clip_ratio/region_mean": 0.01411699817981571, "completions/clipped_ratio": 0.0, "completions/max_length": 180.0, "completions/max_terminated_length": 180.0, "completions/mean_length": 178.25, "completions/mean_terminated_length": 178.25, "completions/min_length": 172.0, "completions/min_terminated_length": 172.0, "entropy": 0.08005253970623016, "epoch": 0.06442543278306623, "frac_reward_zero_std": 0.0, "grad_norm": 1.9800543785095215, "learning_rate": 5.142424242424243e-06, "loss": -0.0024, "num_tokens": 3601733.0, "reward": 0.7351874709129333, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987528324127197, "reward_meter_std": 0.00028449000092223287, "reward_repeat_penalty_mean": 0.7361111044883728, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.0573553666472435, "reward_total_composite_mean": 0.7351874709129333, "reward_total_composite_std": 0.0573553703725338, "reward_total_mean": 0.7351874709129333, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987528324127197, "rewards/meter/std": 0.00028449000092223287, "rewards/repeat_penalty/mean": 0.7361111044883728, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.7351874709129333, "rewards/total_composite/std": 0.0573553703725338, "sampling/importance_sampling_ratio/max": 1.682837963104248, "sampling/importance_sampling_ratio/mean": 1.0028082132339478, "sampling/importance_sampling_ratio/min": 0.18756872415542603, "sampling/sampling_logp_difference/max": 1.673609972000122, "sampling/sampling_logp_difference/mean": 0.01709139533340931, "step": 1604 }, { "clip_ratio/high_max": 0.008729460067115724, "clip_ratio/high_mean": 0.008729460067115724, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.012300888658501208, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.07925032824277878, "epoch": 0.06446559826485118, "frac_reward_zero_std": 0.0, "grad_norm": 0.7806724905967712, "learning_rate": 5.139393939393939e-06, "loss": 0.0008, "num_tokens": 3603520.0, "reward": 0.9987327456474304, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987327456474304, "reward_meter_std": 6.793010834371671e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.793974171159789e-05, "reward_total_composite_mean": 0.9987327456474304, "reward_total_composite_std": 6.793010834371671e-05, "reward_total_mean": 0.9987327456474304, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987327456474304, "rewards/meter/std": 6.793010834371671e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987327456474304, "rewards/total_composite/std": 6.793010834371671e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0033406019210815, "sampling/importance_sampling_ratio/min": 0.3771347403526306, "sampling/sampling_logp_difference/max": 0.9751527309417725, "sampling/sampling_logp_difference/mean": 0.015499016270041466, "step": 1605 }, { "clip_ratio/high_max": 0.0012135922443121672, "clip_ratio/high_mean": 0.0012135922443121672, "clip_ratio/low_mean": 0.0012499999720603228, "clip_ratio/low_min": 0.0012499999720603228, "clip_ratio/region_mean": 0.00246359221637249, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 99.625, "completions/mean_terminated_length": 99.625, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.050112520810216665, "epoch": 0.06450576374663614, "frac_reward_zero_std": 0.0, "grad_norm": 1.1906522512435913, "learning_rate": 5.1363636363636375e-06, "loss": -0.0107, "num_tokens": 3605709.0, "reward": 0.8223729729652405, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9968845248222351, "reward_meter_std": 0.0012903602328151464, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06959936767816544, "reward_total_composite_mean": 0.8223729729652405, "reward_total_composite_std": 0.06959936022758484, "reward_total_mean": 0.8223729729652405, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9968845248222351, "rewards/meter/std": 0.0012903602328151464, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8223729729652405, "rewards/total_composite/std": 0.06959936022758484, "sampling/importance_sampling_ratio/max": 1.2040398120880127, "sampling/importance_sampling_ratio/mean": 1.002079963684082, "sampling/importance_sampling_ratio/min": 0.49163246154785156, "sampling/sampling_logp_difference/max": 0.7100238800048828, "sampling/sampling_logp_difference/mean": 0.006160242948681116, "step": 1606 }, { "clip_ratio/high_max": 0.035856032744050026, "clip_ratio/high_mean": 0.035856032744050026, "clip_ratio/low_mean": 0.010939412750303745, "clip_ratio/low_min": 0.010939412750303745, "clip_ratio/region_mean": 0.04679544549435377, "completions/clipped_ratio": 0.0, "completions/max_length": 183.0, "completions/max_terminated_length": 183.0, "completions/mean_length": 171.75, "completions/mean_terminated_length": 171.75, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.48374373838305473, "epoch": 0.06454592922842109, "frac_reward_zero_std": 0.0, "grad_norm": 4.926582336425781, "learning_rate": 5.133333333333334e-06, "loss": 0.0229, "num_tokens": 3608643.0, "reward": 0.8702612519264221, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.937900185585022, "reward_meter_std": 0.15882940590381622, "reward_repeat_penalty_mean": 0.9319444298744202, "reward_repeat_penalty_std": 0.08195958286523819, "reward_std": 0.15232259035110474, "reward_total_composite_mean": 0.8702612519264221, "reward_total_composite_std": 0.15232259035110474, "reward_total_mean": 0.8702612519264221, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.937900185585022, "rewards/meter/std": 0.15882940590381622, "rewards/repeat_penalty/mean": 0.9319444298744202, "rewards/repeat_penalty/std": 0.08195958286523819, "rewards/total_composite/mean": 0.8702612519264221, "rewards/total_composite/std": 0.15232259035110474, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0150550603866577, "sampling/importance_sampling_ratio/min": 0.2436458021402359, "sampling/sampling_logp_difference/max": 1.4120397567749023, "sampling/sampling_logp_difference/mean": 0.058821044862270355, "step": 1607 }, { "clip_ratio/high_max": 0.014548269449733198, "clip_ratio/high_mean": 0.014548269449733198, "clip_ratio/low_mean": 0.005352868989575654, "clip_ratio/low_min": 0.005352868989575654, "clip_ratio/region_mean": 0.019901138439308852, "completions/clipped_ratio": 0.0, "completions/max_length": 239.0, "completions/max_terminated_length": 239.0, "completions/mean_length": 217.25, "completions/mean_terminated_length": 217.25, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.08922967594116926, "epoch": 0.06458609471020604, "frac_reward_zero_std": 0.0, "grad_norm": 2.1128990650177, "learning_rate": 5.130303030303031e-06, "loss": -0.02, "num_tokens": 3611925.0, "reward": 0.7915375232696533, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9642857313156128, "reward_count_adherence_std": 0.06613000482320786, "reward_meter_mean": 0.9952753782272339, "reward_meter_std": 0.0022598514333367348, "reward_repeat_penalty_mean": 0.8250916004180908, "reward_repeat_penalty_std": 0.05420750379562378, "reward_std": 0.07283391803503036, "reward_total_composite_mean": 0.7915375232696533, "reward_total_composite_std": 0.07283394038677216, "reward_total_mean": 0.7915375232696533, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9642857313156128, "rewards/count_adherence/std": 0.06613000482320786, "rewards/meter/mean": 0.9952753782272339, "rewards/meter/std": 0.0022598514333367348, "rewards/repeat_penalty/mean": 0.8250916004180908, "rewards/repeat_penalty/std": 0.05420750379562378, "rewards/total_composite/mean": 0.7915375232696533, "rewards/total_composite/std": 0.07283394038677216, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.999096691608429, "sampling/importance_sampling_ratio/min": 0.00019008143863175064, "sampling/sampling_logp_difference/max": 8.568058013916016, "sampling/sampling_logp_difference/mean": 0.036408815532922745, "step": 1608 }, { "clip_ratio/high_max": 0.016640397254377604, "clip_ratio/high_mean": 0.016640397254377604, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/region_mean": 0.02073875768110156, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 60.625, "completions/mean_terminated_length": 60.625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.17171061784029007, "epoch": 0.064626260191991, "frac_reward_zero_std": 0.0, "grad_norm": 3.462772846221924, "learning_rate": 5.1272727272727275e-06, "loss": 0.002, "num_tokens": 3613586.0, "reward": 0.9506175518035889, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9920006990432739, "reward_meter_std": 0.00801026076078415, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.11684045940637589, "reward_total_composite_mean": 0.9506175518035889, "reward_total_composite_std": 0.1168404370546341, "reward_total_mean": 0.9506175518035889, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9920006990432739, "rewards/meter/std": 0.00801026076078415, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.9506175518035889, "rewards/total_composite/std": 0.1168404370546341, "sampling/importance_sampling_ratio/max": 1.5062812566757202, "sampling/importance_sampling_ratio/mean": 1.006797194480896, "sampling/importance_sampling_ratio/min": 0.37290093302726746, "sampling/sampling_logp_difference/max": 0.9864425659179688, "sampling/sampling_logp_difference/mean": 0.02973771281540394, "step": 1609 }, { "clip_ratio/high_max": 0.013322061393409967, "clip_ratio/high_mean": 0.013322061393409967, "clip_ratio/low_mean": 0.013008192996494472, "clip_ratio/low_min": 0.013008192996494472, "clip_ratio/region_mean": 0.02633025438990444, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.24806034937500954, "epoch": 0.06466642567377595, "frac_reward_zero_std": 0.0, "grad_norm": 3.0595595836639404, "learning_rate": 5.124242424242425e-06, "loss": 0.0039, "num_tokens": 3615443.0, "reward": 0.9358171224594116, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9358171224594116, "reward_meter_std": 0.018078165128827095, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01807817816734314, "reward_total_composite_mean": 0.9358171224594116, "reward_total_composite_std": 0.018078165128827095, "reward_total_mean": 0.9358171224594116, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9358171224594116, "rewards/meter/std": 0.018078165128827095, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9358171224594116, "rewards/total_composite/std": 0.018078165128827095, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0040435791015625, "sampling/importance_sampling_ratio/min": 0.297671914100647, "sampling/sampling_logp_difference/max": 1.2117633819580078, "sampling/sampling_logp_difference/mean": 0.03475549817085266, "step": 1610 }, { "clip_ratio/high_max": 0.016017435351386666, "clip_ratio/high_mean": 0.016017435351386666, "clip_ratio/low_mean": 0.014760382240638137, "clip_ratio/low_min": 0.014760382240638137, "clip_ratio/region_mean": 0.030777817592024803, "completions/clipped_ratio": 0.0, "completions/max_length": 155.0, "completions/max_terminated_length": 155.0, "completions/mean_length": 141.875, "completions/mean_terminated_length": 141.875, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.5591295659542084, "epoch": 0.0647065911555609, "frac_reward_zero_std": 0.0, "grad_norm": 4.593719959259033, "learning_rate": 5.121212121212121e-06, "loss": 0.0158, "num_tokens": 3617986.0, "reward": 0.86052405834198, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9247380495071411, "reward_meter_std": 0.13655264675617218, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.14909514784812927, "reward_total_composite_mean": 0.86052405834198, "reward_total_composite_std": 0.14909514784812927, "reward_total_mean": 0.86052405834198, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9247380495071411, "rewards/meter/std": 0.13655264675617218, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.86052405834198, "rewards/total_composite/std": 0.14909514784812927, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0124247074127197, "sampling/importance_sampling_ratio/min": 0.0005772316944785416, "sampling/sampling_logp_difference/max": 7.457266807556152, "sampling/sampling_logp_difference/mean": 0.07056324183940887, "step": 1611 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/region_mean": 0.0022727272007614374, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 57.625, "completions/mean_terminated_length": 57.625, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.018967063282616436, "epoch": 0.06474675663734586, "frac_reward_zero_std": 0.0, "grad_norm": 1.593169927597046, "learning_rate": 5.1181818181818185e-06, "loss": -0.0037, "num_tokens": 3619751.0, "reward": 0.9934520721435547, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9934520721435547, "reward_meter_std": 3.5445533285383135e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.5451525036478415e-05, "reward_total_composite_mean": 0.9934520721435547, "reward_total_composite_std": 3.5445533285383135e-05, "reward_total_mean": 0.9934520721435547, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9934520721435547, "rewards/meter/std": 3.5445533285383135e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934520721435547, "rewards/total_composite/std": 3.5445533285383135e-05, "sampling/importance_sampling_ratio/max": 1.5921474695205688, "sampling/importance_sampling_ratio/mean": 1.0014033317565918, "sampling/importance_sampling_ratio/min": 0.6644558906555176, "sampling/sampling_logp_difference/max": 0.46508365869522095, "sampling/sampling_logp_difference/mean": 0.003705526702105999, "step": 1612 }, { "clip_ratio/high_max": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.001953125, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 63.75, "completions/mean_terminated_length": 63.75, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.056046845857053995, "epoch": 0.06478692211913081, "frac_reward_zero_std": 0.0, "grad_norm": 7.702324390411377, "learning_rate": 5.115151515151515e-06, "loss": -0.0043, "num_tokens": 3621517.0, "reward": 0.9974002838134766, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974002838134766, "reward_meter_std": 0.0021041962318122387, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002104192739352584, "reward_total_composite_mean": 0.9974002838134766, "reward_total_composite_std": 0.0021041962318122387, "reward_total_mean": 0.9974002838134766, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974002838134766, "rewards/meter/std": 0.0021041962318122387, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974002838134766, "rewards/total_composite/std": 0.0021041962318122387, "sampling/importance_sampling_ratio/max": 1.317487359046936, "sampling/importance_sampling_ratio/mean": 1.0017887353897095, "sampling/importance_sampling_ratio/min": 0.46585214138031006, "sampling/sampling_logp_difference/max": 0.7638870477676392, "sampling/sampling_logp_difference/mean": 0.01148869190365076, "step": 1613 }, { "clip_ratio/high_max": 0.02137809677515179, "clip_ratio/high_mean": 0.02137809677515179, "clip_ratio/low_mean": 0.018685899674892426, "clip_ratio/low_min": 0.018685899674892426, "clip_ratio/region_mean": 0.040063996450044215, "completions/clipped_ratio": 0.0, "completions/max_length": 198.0, "completions/max_terminated_length": 198.0, "completions/mean_length": 169.5, "completions/mean_terminated_length": 169.5, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.3938692733645439, "epoch": 0.06482708760091577, "frac_reward_zero_std": 0.0, "grad_norm": 3.499596118927002, "learning_rate": 5.112121212121213e-06, "loss": 0.0583, "num_tokens": 3624361.0, "reward": 0.6842913627624512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_meter_mean": 0.9390597343444824, "reward_meter_std": 0.12540623545646667, "reward_repeat_penalty_mean": 0.7876983880996704, "reward_repeat_penalty_std": 0.13440276682376862, "reward_std": 0.13273297250270844, "reward_total_composite_mean": 0.6842913627624512, "reward_total_composite_std": 0.13273297250270844, "reward_total_mean": 0.6842913627624512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/meter/mean": 0.9390597343444824, "rewards/meter/std": 0.12540623545646667, "rewards/repeat_penalty/mean": 0.7876983880996704, "rewards/repeat_penalty/std": 0.13440276682376862, "rewards/total_composite/mean": 0.6842913627624512, "rewards/total_composite/std": 0.13273297250270844, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.012567162513733, "sampling/importance_sampling_ratio/min": 0.020648617297410965, "sampling/sampling_logp_difference/max": 3.8801069259643555, "sampling/sampling_logp_difference/mean": 0.05148252472281456, "step": 1614 }, { "clip_ratio/high_max": 0.006465517217293382, "clip_ratio/high_mean": 0.006465517217293382, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006465517217293382, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.02835727483034134, "epoch": 0.06486725308270072, "frac_reward_zero_std": 0.0, "grad_norm": 9.031867027282715, "learning_rate": 5.109090909090909e-06, "loss": 0.0083, "num_tokens": 3626009.0, "reward": 0.992936372756958, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.992936372756958, "reward_meter_std": 0.0014942112611606717, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0014942261623218656, "reward_total_composite_mean": 0.992936372756958, "reward_total_composite_std": 0.0014942112611606717, "reward_total_mean": 0.992936372756958, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.992936372756958, "rewards/meter/std": 0.0014942112611606717, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992936372756958, "rewards/total_composite/std": 0.0014942112611606717, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016261339187622, "sampling/importance_sampling_ratio/min": 0.5572132468223572, "sampling/sampling_logp_difference/max": 0.7060290575027466, "sampling/sampling_logp_difference/mean": 0.008410189300775528, "step": 1615 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/region_mean": 0.0017361111240461469, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.07067765574902296, "epoch": 0.06490741856448567, "frac_reward_zero_std": 0.0, "grad_norm": 3.926830530166626, "learning_rate": 5.106060606060607e-06, "loss": 0.02, "num_tokens": 3627780.0, "reward": 0.9946064352989197, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946064352989197, "reward_meter_std": 0.0011085572186857462, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0011085484875366092, "reward_total_composite_mean": 0.9946064352989197, "reward_total_composite_std": 0.0011085572186857462, "reward_total_mean": 0.9946064352989197, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946064352989197, "rewards/meter/std": 0.0011085572186857462, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946064352989197, "rewards/total_composite/std": 0.0011085572186857462, "sampling/importance_sampling_ratio/max": 1.385042428970337, "sampling/importance_sampling_ratio/mean": 1.0016788244247437, "sampling/importance_sampling_ratio/min": 0.4407116174697876, "sampling/sampling_logp_difference/max": 0.8193645477294922, "sampling/sampling_logp_difference/mean": 0.009593191556632519, "step": 1616 }, { "clip_ratio/high_max": 0.023898042272776365, "clip_ratio/high_mean": 0.023898042272776365, "clip_ratio/low_mean": 0.003343421034514904, "clip_ratio/low_min": 0.003343421034514904, "clip_ratio/region_mean": 0.02724146330729127, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 445.0, "completions/mean_terminated_length": 435.4285888671875, "completions/min_length": 421.0, "completions/min_terminated_length": 421.0, "entropy": 0.30827769078314304, "epoch": 0.06494758404627063, "frac_reward_zero_std": 0.0, "grad_norm": 1.3416348695755005, "learning_rate": 5.103030303030303e-06, "loss": -0.1647, "num_tokens": 3633068.0, "reward": 0.4080624282360077, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.671875, "reward_count_adherence_std": 0.07281029969453812, "reward_meter_mean": 0.9931340217590332, "reward_meter_std": 0.005852298345416784, "reward_repeat_penalty_mean": 0.7910888195037842, "reward_repeat_penalty_std": 0.1950831413269043, "reward_std": 0.282764196395874, "reward_total_composite_mean": 0.4080624282360077, "reward_total_composite_std": 0.282764196395874, "reward_total_mean": 0.4080624282360077, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.671875, "rewards/count_adherence/std": 0.07281029969453812, "rewards/meter/mean": 0.9931340217590332, "rewards/meter/std": 0.005852298345416784, "rewards/repeat_penalty/mean": 0.7910888195037842, "rewards/repeat_penalty/std": 0.1950831413269043, "rewards/total_composite/mean": 0.4080624282360077, "rewards/total_composite/std": 0.282764196395874, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0104022026062012, "sampling/importance_sampling_ratio/min": 0.02755417674779892, "sampling/sampling_logp_difference/max": 3.5916011333465576, "sampling/sampling_logp_difference/mean": 0.04456779360771179, "step": 1617 }, { "clip_ratio/high_max": 0.02332985820248723, "clip_ratio/high_mean": 0.02332985820248723, "clip_ratio/low_mean": 0.009485365822911263, "clip_ratio/low_min": 0.009485365822911263, "clip_ratio/region_mean": 0.03281522402539849, "completions/clipped_ratio": 0.0, "completions/max_length": 402.0, "completions/max_terminated_length": 402.0, "completions/mean_length": 389.5, "completions/mean_terminated_length": 389.5, "completions/min_length": 376.0, "completions/min_terminated_length": 376.0, "entropy": 0.3395629785954952, "epoch": 0.06498774952805558, "frac_reward_zero_std": 0.0, "grad_norm": 1.6642099618911743, "learning_rate": 5.1e-06, "loss": -0.0118, "num_tokens": 3637584.0, "reward": 0.597885012626648, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942715167999268, "reward_meter_std": 0.005808908957988024, "reward_repeat_penalty_mean": 0.8421052694320679, "reward_repeat_penalty_std": 0.14886459708213806, "reward_std": 0.10485603660345078, "reward_total_composite_mean": 0.597885012626648, "reward_total_composite_std": 0.10485604405403137, "reward_total_mean": 0.597885012626648, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942715167999268, "rewards/meter/std": 0.005808908957988024, "rewards/repeat_penalty/mean": 0.8421052694320679, "rewards/repeat_penalty/std": 0.14886459708213806, "rewards/total_composite/mean": 0.597885012626648, "rewards/total_composite/std": 0.10485604405403137, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0039002895355225, "sampling/importance_sampling_ratio/min": 0.08708232641220093, "sampling/sampling_logp_difference/max": 2.440901279449463, "sampling/sampling_logp_difference/mean": 0.044363927096128464, "step": 1618 }, { "clip_ratio/high_max": 0.004385964944958687, "clip_ratio/high_mean": 0.004385964944958687, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/region_mean": 0.006504609016701579, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 59.0, "completions/mean_terminated_length": 59.0, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.12759912386536598, "epoch": 0.06502791500984054, "frac_reward_zero_std": 0.0, "grad_norm": 10.757038116455078, "learning_rate": 5.096969696969697e-06, "loss": 0.0001, "num_tokens": 3639232.0, "reward": 0.9416102766990662, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9416102766990662, "reward_meter_std": 0.15220597386360168, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1522059589624405, "reward_total_composite_mean": 0.9416102766990662, "reward_total_composite_std": 0.15220597386360168, "reward_total_mean": 0.9416102766990662, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9416102766990662, "rewards/meter/std": 0.15220597386360168, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9416102766990662, "rewards/total_composite/std": 0.15220597386360168, "sampling/importance_sampling_ratio/max": 1.6727921962738037, "sampling/importance_sampling_ratio/mean": 1.007680892944336, "sampling/importance_sampling_ratio/min": 0.5884917974472046, "sampling/sampling_logp_difference/max": 0.5301923751831055, "sampling/sampling_logp_difference/mean": 0.018271297216415405, "step": 1619 }, { "clip_ratio/high_max": 0.02794372313655913, "clip_ratio/high_mean": 0.02794372313655913, "clip_ratio/low_mean": 0.01416793093085289, "clip_ratio/low_min": 0.01416793093085289, "clip_ratio/region_mean": 0.04211165406741202, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 77.875, "completions/mean_terminated_length": 77.875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.5332462079823017, "epoch": 0.06506808049162549, "frac_reward_zero_std": 0.0, "grad_norm": 4.442444801330566, "learning_rate": 5.093939393939395e-06, "loss": 0.032, "num_tokens": 3641223.0, "reward": 0.9966658353805542, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9966658353805542, "reward_meter_std": 0.0016763400053605437, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0016763500170782208, "reward_total_composite_mean": 0.9966658353805542, "reward_total_composite_std": 0.0016763400053605437, "reward_total_mean": 0.9966658353805542, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9966658353805542, "rewards/meter/std": 0.0016763400053605437, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9966658353805542, "rewards/total_composite/std": 0.0016763400053605437, "sampling/importance_sampling_ratio/max": 1.952990174293518, "sampling/importance_sampling_ratio/mean": 1.0116773843765259, "sampling/importance_sampling_ratio/min": 0.23465938866138458, "sampling/sampling_logp_difference/max": 1.449620246887207, "sampling/sampling_logp_difference/mean": 0.06199686974287033, "step": 1620 }, { "clip_ratio/high_max": 0.0513387790415436, "clip_ratio/high_mean": 0.0513387790415436, "clip_ratio/low_mean": 0.02026098920032382, "clip_ratio/low_min": 0.02026098920032382, "clip_ratio/region_mean": 0.07159976824186742, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 46.5, "completions/mean_terminated_length": 46.5, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.7850339710712433, "epoch": 0.06510824597341044, "frac_reward_zero_std": 0.0, "grad_norm": 9.210960388183594, "learning_rate": 5.090909090909091e-06, "loss": 0.2585, "num_tokens": 3642899.0, "reward": 0.959256649017334, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.959256649017334, "reward_meter_std": 0.07027505338191986, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.07027503848075867, "reward_total_composite_mean": 0.959256649017334, "reward_total_composite_std": 0.07027505338191986, "reward_total_mean": 0.959256649017334, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.959256649017334, "rewards/meter/std": 0.07027505338191986, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.959256649017334, "rewards/total_composite/std": 0.07027505338191986, "sampling/importance_sampling_ratio/max": 1.8360445499420166, "sampling/importance_sampling_ratio/mean": 1.0092580318450928, "sampling/importance_sampling_ratio/min": 0.2843836545944214, "sampling/sampling_logp_difference/max": 1.2574310302734375, "sampling/sampling_logp_difference/mean": 0.0790342465043068, "step": 1621 }, { "clip_ratio/high_max": 0.01896783267147839, "clip_ratio/high_mean": 0.01896783267147839, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.01896783267147839, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 55.625, "completions/mean_terminated_length": 55.625, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.16878793388605118, "epoch": 0.0651484114551954, "frac_reward_zero_std": 0.0, "grad_norm": 5.405808925628662, "learning_rate": 5.0878787878787885e-06, "loss": -0.1515, "num_tokens": 3644592.0, "reward": 0.9314955472946167, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.993726372718811, "reward_meter_std": 0.0034743899013847113, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.17525316774845123, "reward_total_composite_mean": 0.9314955472946167, "reward_total_composite_std": 0.17525316774845123, "reward_total_mean": 0.9314955472946167, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.993726372718811, "rewards/meter/std": 0.0034743899013847113, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9314955472946167, "rewards/total_composite/std": 0.17525316774845123, "sampling/importance_sampling_ratio/max": 1.6441330909729004, "sampling/importance_sampling_ratio/mean": 1.0055111646652222, "sampling/importance_sampling_ratio/min": 0.3641340732574463, "sampling/sampling_logp_difference/max": 1.0102331638336182, "sampling/sampling_logp_difference/mean": 0.02844316139817238, "step": 1622 }, { "clip_ratio/high_max": 0.04867990920320153, "clip_ratio/high_mean": 0.04867990920320153, "clip_ratio/low_mean": 0.0051369862630963326, "clip_ratio/low_min": 0.0051369862630963326, "clip_ratio/region_mean": 0.053816895466297865, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.0, "completions/mean_terminated_length": 72.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.6054626666009426, "epoch": 0.06518857693698035, "frac_reward_zero_std": 0.0, "grad_norm": 4.352970123291016, "learning_rate": 5.084848484848486e-06, "loss": 0.0053, "num_tokens": 3646304.0, "reward": 0.8835433125495911, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8835433125495911, "reward_meter_std": 0.2843259274959564, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2843259274959564, "reward_total_composite_mean": 0.8835433125495911, "reward_total_composite_std": 0.2843259274959564, "reward_total_mean": 0.8835433125495911, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8835433125495911, "rewards/meter/std": 0.2843259274959564, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8835433125495911, "rewards/total_composite/std": 0.2843259274959564, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0209115743637085, "sampling/importance_sampling_ratio/min": 0.1102001741528511, "sampling/sampling_logp_difference/max": 2.2054567337036133, "sampling/sampling_logp_difference/mean": 0.061763398349285126, "step": 1623 }, { "clip_ratio/high_max": 0.022054610773921013, "clip_ratio/high_mean": 0.022054610773921013, "clip_ratio/low_mean": 0.016402944223955274, "clip_ratio/low_min": 0.016402944223955274, "clip_ratio/region_mean": 0.038457554997876287, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 336.0, "completions/mean_terminated_length": 336.0, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "entropy": 0.5225833132863045, "epoch": 0.0652287424187653, "frac_reward_zero_std": 0.0, "grad_norm": 2.14442777633667, "learning_rate": 5.081818181818182e-06, "loss": -0.0003, "num_tokens": 3650768.0, "reward": 0.8190490007400513, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.046290989965200424, "reward_meter_mean": 0.9964303374290466, "reward_meter_std": 0.0028279528487473726, "reward_repeat_penalty_mean": 0.9401960372924805, "reward_repeat_penalty_std": 0.04455278813838959, "reward_std": 0.04663977771997452, "reward_total_composite_mean": 0.8190490007400513, "reward_total_composite_std": 0.04663979262113571, "reward_total_mean": 0.8190490007400513, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.046290989965200424, "rewards/meter/mean": 0.9964303374290466, "rewards/meter/std": 0.0028279528487473726, "rewards/repeat_penalty/mean": 0.9401960372924805, "rewards/repeat_penalty/std": 0.04455278813838959, "rewards/total_composite/mean": 0.8190490007400513, "rewards/total_composite/std": 0.04663979262113571, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01429283618927, "sampling/importance_sampling_ratio/min": 0.24766594171524048, "sampling/sampling_logp_difference/max": 1.395674467086792, "sampling/sampling_logp_difference/mean": 0.058607399463653564, "step": 1624 }, { "clip_ratio/high_max": 0.04015351925045252, "clip_ratio/high_mean": 0.04015351925045252, "clip_ratio/low_mean": 0.02405611891299486, "clip_ratio/low_min": 0.02405611891299486, "clip_ratio/region_mean": 0.06420963816344738, "completions/clipped_ratio": 0.0, "completions/max_length": 189.0, "completions/max_terminated_length": 189.0, "completions/mean_length": 171.375, "completions/mean_terminated_length": 171.375, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.7438356392085552, "epoch": 0.06526890790055026, "frac_reward_zero_std": 0.0, "grad_norm": 4.290684223175049, "learning_rate": 5.078787878787879e-06, "loss": -0.0375, "num_tokens": 3653643.0, "reward": 0.8372185230255127, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_meter_mean": 0.9575715065002441, "reward_meter_std": 0.10626022517681122, "reward_repeat_penalty_mean": 0.9682539701461792, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.15995362401008606, "reward_total_composite_mean": 0.8372185230255127, "reward_total_composite_std": 0.15995363891124725, "reward_total_mean": 0.8372185230255127, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/meter/mean": 0.9575715065002441, "rewards/meter/std": 0.10626022517681122, "rewards/repeat_penalty/mean": 0.9682539701461792, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.8372185230255127, "rewards/total_composite/std": 0.15995363891124725, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.009111762046814, "sampling/importance_sampling_ratio/min": 0.037647999823093414, "sampling/sampling_logp_difference/max": 3.279475450515747, "sampling/sampling_logp_difference/mean": 0.09049484878778458, "step": 1625 }, { "clip_ratio/high_max": 0.01032647315878421, "clip_ratio/high_mean": 0.01032647315878421, "clip_ratio/low_mean": 0.017135416390374303, "clip_ratio/low_min": 0.017135416390374303, "clip_ratio/region_mean": 0.027461889549158514, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 96.5, "completions/mean_terminated_length": 96.5, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.45825108513236046, "epoch": 0.06530907338233521, "frac_reward_zero_std": 0.0, "grad_norm": 3.822707414627075, "learning_rate": 5.075757575757576e-06, "loss": 0.0078, "num_tokens": 3655935.0, "reward": 0.9368202686309814, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9368202686309814, "reward_meter_std": 0.026791444048285484, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02679145336151123, "reward_total_composite_mean": 0.9368202686309814, "reward_total_composite_std": 0.026791444048285484, "reward_total_mean": 0.9368202686309814, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9368202686309814, "rewards/meter/std": 0.026791444048285484, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9368202686309814, "rewards/total_composite/std": 0.026791444048285484, "sampling/importance_sampling_ratio/max": 1.796831727027893, "sampling/importance_sampling_ratio/mean": 1.0122920274734497, "sampling/importance_sampling_ratio/min": 0.37992405891418457, "sampling/sampling_logp_difference/max": 0.9677839279174805, "sampling/sampling_logp_difference/mean": 0.05230702832341194, "step": 1626 }, { "clip_ratio/high_max": 0.014909721445292234, "clip_ratio/high_mean": 0.014909721445292234, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/region_mean": 0.01900808187201619, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 59.375, "completions/mean_terminated_length": 59.375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.1392963044345379, "epoch": 0.06534923886412017, "frac_reward_zero_std": 0.0, "grad_norm": 6.298683166503906, "learning_rate": 5.072727272727274e-06, "loss": 0.0042, "num_tokens": 3657666.0, "reward": 0.8852929472923279, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8852929472923279, "reward_meter_std": 0.30933257937431335, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.30933254957199097, "reward_total_composite_mean": 0.8852929472923279, "reward_total_composite_std": 0.30933257937431335, "reward_total_mean": 0.8852929472923279, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8852929472923279, "rewards/meter/std": 0.30933257937431335, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8852929472923279, "rewards/total_composite/std": 0.30933257937431335, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0004394054412842, "sampling/importance_sampling_ratio/min": 0.37357792258262634, "sampling/sampling_logp_difference/max": 0.9846286773681641, "sampling/sampling_logp_difference/mean": 0.025638654828071594, "step": 1627 }, { "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/region_mean": 0.006949807051569223, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 35.25, "completions/mean_terminated_length": 35.25, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.060785280307754874, "epoch": 0.06538940434590514, "frac_reward_zero_std": 0.0, "grad_norm": 3.436232328414917, "learning_rate": 5.06969696969697e-06, "loss": 0.0099, "num_tokens": 3659244.0, "reward": 0.9971714019775391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971714019775391, "reward_meter_std": 0.0016039953334257007, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0016039892798289657, "reward_total_composite_mean": 0.9971714019775391, "reward_total_composite_std": 0.0016039953334257007, "reward_total_mean": 0.9971714019775391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971714019775391, "rewards/meter/std": 0.0016039953334257007, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971714019775391, "rewards/total_composite/std": 0.0016039953334257007, "sampling/importance_sampling_ratio/max": 1.063112735748291, "sampling/importance_sampling_ratio/mean": 0.9952727556228638, "sampling/importance_sampling_ratio/min": 0.4319186508655548, "sampling/sampling_logp_difference/max": 0.8395180702209473, "sampling/sampling_logp_difference/mean": 0.0144999660551548, "step": 1628 }, { "clip_ratio/high_max": 0.03499942785128951, "clip_ratio/high_mean": 0.03499942785128951, "clip_ratio/low_mean": 0.004980936297215521, "clip_ratio/low_min": 0.004980936297215521, "clip_ratio/region_mean": 0.03998036414850503, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 339.125, "completions/mean_terminated_length": 339.125, "completions/min_length": 321.0, "completions/min_terminated_length": 321.0, "entropy": 0.622920136898756, "epoch": 0.06542956982769009, "frac_reward_zero_std": 0.0, "grad_norm": 3.441396474838257, "learning_rate": 5.0666666666666676e-06, "loss": -0.0053, "num_tokens": 3663869.0, "reward": 0.7674725651741028, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8068181276321411, "reward_count_adherence_std": 0.03214120864868164, "reward_meter_mean": 0.9954209923744202, "reward_meter_std": 0.0042012035846710205, "reward_repeat_penalty_mean": 0.9562908411026001, "reward_repeat_penalty_std": 0.060786258429288864, "reward_std": 0.04974529147148132, "reward_total_composite_mean": 0.7674725651741028, "reward_total_composite_std": 0.04974528029561043, "reward_total_mean": 0.7674725651741028, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8068181276321411, "rewards/count_adherence/std": 0.03214120864868164, "rewards/meter/mean": 0.9954209923744202, "rewards/meter/std": 0.0042012035846710205, "rewards/repeat_penalty/mean": 0.9562908411026001, "rewards/repeat_penalty/std": 0.060786258429288864, "rewards/total_composite/mean": 0.7674725651741028, "rewards/total_composite/std": 0.04974528029561043, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0179475545883179, "sampling/importance_sampling_ratio/min": 0.2069363296031952, "sampling/sampling_logp_difference/max": 1.5753440856933594, "sampling/sampling_logp_difference/mean": 0.06467485427856445, "step": 1629 }, { "clip_ratio/high_max": 0.02247373666614294, "clip_ratio/high_mean": 0.02247373666614294, "clip_ratio/low_mean": 0.031975225545465946, "clip_ratio/low_min": 0.031975225545465946, "clip_ratio/region_mean": 0.05444896221160889, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 343.125, "completions/mean_terminated_length": 343.125, "completions/min_length": 326.0, "completions/min_terminated_length": 326.0, "entropy": 0.7744914256036282, "epoch": 0.06546973530947504, "frac_reward_zero_std": 0.0, "grad_norm": 2.8746509552001953, "learning_rate": 5.063636363636364e-06, "loss": 0.01, "num_tokens": 3668262.0, "reward": 0.8051279783248901, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.862500011920929, "reward_count_adherence_std": 0.0517548993229866, "reward_meter_mean": 0.9573882222175598, "reward_meter_std": 0.10380185395479202, "reward_repeat_penalty_mean": 0.9774816036224365, "reward_repeat_penalty_std": 0.04422420263290405, "reward_std": 0.09284677356481552, "reward_total_composite_mean": 0.8051279783248901, "reward_total_composite_std": 0.09284678101539612, "reward_total_mean": 0.8051279783248901, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.862500011920929, "rewards/count_adherence/std": 0.0517548993229866, "rewards/meter/mean": 0.9573882222175598, "rewards/meter/std": 0.10380185395479202, "rewards/repeat_penalty/mean": 0.9774816036224365, "rewards/repeat_penalty/std": 0.04422420263290405, "rewards/total_composite/mean": 0.8051279783248901, "rewards/total_composite/std": 0.09284678101539612, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.016446590423584, "sampling/importance_sampling_ratio/min": 0.17942388355731964, "sampling/sampling_logp_difference/max": 1.7180042266845703, "sampling/sampling_logp_difference/mean": 0.07694988697767258, "step": 1630 }, { "clip_ratio/high_max": 0.006843626964837313, "clip_ratio/high_mean": 0.006843626964837313, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.010749876964837313, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 127.875, "completions/mean_terminated_length": 127.875, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.02229404100216925, "epoch": 0.06550990079126, "frac_reward_zero_std": 0.0, "grad_norm": 0.9008557200431824, "learning_rate": 5.060606060606061e-06, "loss": 0.0001, "num_tokens": 3670677.0, "reward": 0.5963674187660217, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939456582069397, "reward_meter_std": 0.00018129698582924902, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001087786877178587, "reward_total_composite_mean": 0.5963674187660217, "reward_total_composite_std": 0.00010878406465053558, "reward_total_mean": 0.5963674187660217, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939456582069397, "rewards/meter/std": 0.00018129698582924902, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5963674187660217, "rewards/total_composite/std": 0.00010878406465053558, "sampling/importance_sampling_ratio/max": 1.6699923276901245, "sampling/importance_sampling_ratio/mean": 1.0000206232070923, "sampling/importance_sampling_ratio/min": 0.41602447628974915, "sampling/sampling_logp_difference/max": 0.8770111799240112, "sampling/sampling_logp_difference/mean": 0.007339623291045427, "step": 1631 }, { "clip_ratio/high_max": 0.04253894090652466, "clip_ratio/high_mean": 0.04253894090652466, "clip_ratio/low_mean": 0.027231683605350554, "clip_ratio/low_min": 0.027231683605350554, "clip_ratio/region_mean": 0.06977062451187521, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 73.125, "completions/mean_terminated_length": 73.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.7365500181913376, "epoch": 0.06555006627304495, "frac_reward_zero_std": 0.0, "grad_norm": 4.735329627990723, "learning_rate": 5.057575757575758e-06, "loss": -0.0158, "num_tokens": 3672574.0, "reward": 0.9905960559844971, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9905960559844971, "reward_meter_std": 0.008815782144665718, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008815782144665718, "reward_total_composite_mean": 0.9905960559844971, "reward_total_composite_std": 0.008815782144665718, "reward_total_mean": 0.9905960559844971, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9905960559844971, "rewards/meter/std": 0.008815782144665718, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9905960559844971, "rewards/total_composite/std": 0.008815782144665718, "sampling/importance_sampling_ratio/max": 1.9160219430923462, "sampling/importance_sampling_ratio/mean": 1.0182366371154785, "sampling/importance_sampling_ratio/min": 0.3070189952850342, "sampling/sampling_logp_difference/max": 1.1808457374572754, "sampling/sampling_logp_difference/mean": 0.07114405184984207, "step": 1632 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 30.0, "completions/mean_terminated_length": 30.0, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.004795511835254729, "epoch": 0.0655902317548299, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.054545454545455e-06, "loss": 0.0, "num_tokens": 3674102.0, "reward": 0.992271900177002, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.992271900177002, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.992271900177002, "reward_total_composite_std": 0.0, "reward_total_mean": 0.992271900177002, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.992271900177002, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992271900177002, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0058854818344116, "sampling/importance_sampling_ratio/mean": 1.0005319118499756, "sampling/importance_sampling_ratio/min": 0.998640239238739, "sampling/sampling_logp_difference/max": 0.005868114531040192, "sampling/sampling_logp_difference/mean": 0.0005566918407566845, "step": 1633 }, { "clip_ratio/high_max": 0.03815131215378642, "clip_ratio/high_mean": 0.03815131215378642, "clip_ratio/low_mean": 0.013648775406181812, "clip_ratio/low_min": 0.013648775406181812, "clip_ratio/region_mean": 0.05180008755996823, "completions/clipped_ratio": 0.0, "completions/max_length": 113.0, "completions/max_terminated_length": 113.0, "completions/mean_length": 108.625, "completions/mean_terminated_length": 108.625, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.6268335022032261, "epoch": 0.06563039723661486, "frac_reward_zero_std": 0.0, "grad_norm": 5.207538604736328, "learning_rate": 5.051515151515151e-06, "loss": 0.0068, "num_tokens": 3676259.0, "reward": 0.9962839484214783, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962839484214783, "reward_meter_std": 0.0024302289821207523, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0024302229285240173, "reward_total_composite_mean": 0.9962839484214783, "reward_total_composite_std": 0.0024302289821207523, "reward_total_mean": 0.9962839484214783, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962839484214783, "rewards/meter/std": 0.0024302289821207523, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9962839484214783, "rewards/total_composite/std": 0.0024302289821207523, "sampling/importance_sampling_ratio/max": 1.8952765464782715, "sampling/importance_sampling_ratio/mean": 1.0127630233764648, "sampling/importance_sampling_ratio/min": 0.2423473745584488, "sampling/sampling_logp_difference/max": 1.4173831939697266, "sampling/sampling_logp_difference/mean": 0.0648764818906784, "step": 1634 }, { "clip_ratio/high_max": 0.042927493108436465, "clip_ratio/high_mean": 0.042927493108436465, "clip_ratio/low_mean": 0.01331305643543601, "clip_ratio/low_min": 0.01331305643543601, "clip_ratio/region_mean": 0.056240549543872476, "completions/clipped_ratio": 0.0, "completions/max_length": 166.0, "completions/max_terminated_length": 166.0, "completions/mean_length": 160.125, "completions/mean_terminated_length": 160.125, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "entropy": 0.9452630281448364, "epoch": 0.06567056271839981, "frac_reward_zero_std": 0.0, "grad_norm": 4.150503158569336, "learning_rate": 5.048484848484849e-06, "loss": -0.0025, "num_tokens": 3679020.0, "reward": 0.974323570728302, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9920204877853394, "reward_meter_std": 0.01226998120546341, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05192466452717781, "reward_total_composite_mean": 0.974323570728302, "reward_total_composite_std": 0.05192466080188751, "reward_total_mean": 0.974323570728302, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9920204877853394, "rewards/meter/std": 0.01226998120546341, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.974323570728302, "rewards/total_composite/std": 0.05192466080188751, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0256816148757935, "sampling/importance_sampling_ratio/min": 0.2856857180595398, "sampling/sampling_logp_difference/max": 1.2528629302978516, "sampling/sampling_logp_difference/mean": 0.08837065100669861, "step": 1635 }, { "clip_ratio/high_max": 0.00390625, "clip_ratio/high_mean": 0.00390625, "clip_ratio/low_mean": 0.007905506063252687, "clip_ratio/low_min": 0.007905506063252687, "clip_ratio/region_mean": 0.011811756063252687, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 63.875, "completions/mean_terminated_length": 63.875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.09926960244774818, "epoch": 0.06571072820018477, "frac_reward_zero_std": 0.0, "grad_norm": 2.887805223464966, "learning_rate": 5.045454545454546e-06, "loss": 0.0004, "num_tokens": 3680811.0, "reward": 0.9980711936950684, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980711936950684, "reward_meter_std": 0.0005036595975980163, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005036554648540914, "reward_total_composite_mean": 0.9980711936950684, "reward_total_composite_std": 0.0005036595975980163, "reward_total_mean": 0.9980711936950684, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980711936950684, "rewards/meter/std": 0.0005036595975980163, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980711936950684, "rewards/total_composite/std": 0.0005036595975980163, "sampling/importance_sampling_ratio/max": 1.4901469945907593, "sampling/importance_sampling_ratio/mean": 0.9980214238166809, "sampling/importance_sampling_ratio/min": 0.477216899394989, "sampling/sampling_logp_difference/max": 0.7397842407226562, "sampling/sampling_logp_difference/mean": 0.019482869654893875, "step": 1636 }, { "clip_ratio/high_max": 0.05837191268801689, "clip_ratio/high_mean": 0.05837191268801689, "clip_ratio/low_mean": 0.026575686410069466, "clip_ratio/low_min": 0.026575686410069466, "clip_ratio/region_mean": 0.08494759909808636, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 69.875, "completions/mean_terminated_length": 69.875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 1.0178619101643562, "epoch": 0.06575089368196972, "frac_reward_zero_std": 0.0, "grad_norm": 7.507004737854004, "learning_rate": 5.042424242424243e-06, "loss": 0.0281, "num_tokens": 3682330.0, "reward": 0.6862971782684326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6862971782684326, "reward_meter_std": 0.2807905673980713, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2807905673980713, "reward_total_composite_mean": 0.6862971782684326, "reward_total_composite_std": 0.2807905673980713, "reward_total_mean": 0.6862971782684326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6862971782684326, "rewards/meter/std": 0.2807905673980713, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6862971782684326, "rewards/total_composite/std": 0.2807905673980713, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0039104223251343, "sampling/importance_sampling_ratio/min": 0.22169609367847443, "sampling/sampling_logp_difference/max": 1.5064477920532227, "sampling/sampling_logp_difference/mean": 0.10850861668586731, "step": 1637 }, { "clip_ratio/high_max": 0.010610593715682626, "clip_ratio/high_mean": 0.010610593715682626, "clip_ratio/low_mean": 0.006431827903725207, "clip_ratio/low_min": 0.006431827903725207, "clip_ratio/region_mean": 0.017042421619407833, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 95.0, "completions/mean_terminated_length": 95.0, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.10523929260671139, "epoch": 0.06579105916375468, "frac_reward_zero_std": 0.0, "grad_norm": 3.420053720474243, "learning_rate": 5.0393939393939395e-06, "loss": 0.0097, "num_tokens": 3684458.0, "reward": 0.8965927362442017, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959671497344971, "reward_meter_std": 0.006246180739253759, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.10870993137359619, "reward_total_composite_mean": 0.8965927362442017, "reward_total_composite_std": 0.1087099239230156, "reward_total_mean": 0.8965927362442017, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959671497344971, "rewards/meter/std": 0.006246180739253759, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8965927362442017, "rewards/total_composite/std": 0.1087099239230156, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0042842626571655, "sampling/importance_sampling_ratio/min": 0.6205868124961853, "sampling/sampling_logp_difference/max": 0.9803454875946045, "sampling/sampling_logp_difference/mean": 0.01706601306796074, "step": 1638 }, { "clip_ratio/high_max": 0.005859375, "clip_ratio/high_mean": 0.005859375, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.009765625, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.06925144325941801, "epoch": 0.06583122464553963, "frac_reward_zero_std": 0.0, "grad_norm": 1.6942023038864136, "learning_rate": 5.036363636363637e-06, "loss": 0.0003, "num_tokens": 3686354.0, "reward": 0.9983768463134766, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983768463134766, "reward_meter_std": 0.00010835672583198175, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010834616841748357, "reward_total_composite_mean": 0.9983768463134766, "reward_total_composite_std": 0.00010835672583198175, "reward_total_mean": 0.9983768463134766, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983768463134766, "rewards/meter/std": 0.00010835672583198175, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983768463134766, "rewards/total_composite/std": 0.00010835672583198175, "sampling/importance_sampling_ratio/max": 1.363644003868103, "sampling/importance_sampling_ratio/mean": 1.0026438236236572, "sampling/importance_sampling_ratio/min": 0.434855192899704, "sampling/sampling_logp_difference/max": 0.8327422142028809, "sampling/sampling_logp_difference/mean": 0.01173328422009945, "step": 1639 }, { "clip_ratio/high_max": 0.012254902278073132, "clip_ratio/high_mean": 0.012254902278073132, "clip_ratio/low_mean": 0.010499144555069506, "clip_ratio/low_min": 0.010499144555069506, "clip_ratio/region_mean": 0.022754046833142638, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 69.75, "completions/mean_terminated_length": 69.75, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.2406612578779459, "epoch": 0.06587139012732458, "frac_reward_zero_std": 0.0, "grad_norm": 5.066521167755127, "learning_rate": 5.033333333333333e-06, "loss": 0.0239, "num_tokens": 3688136.0, "reward": 0.9961869716644287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961869716644287, "reward_meter_std": 0.0016419171588495374, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0016419151797890663, "reward_total_composite_mean": 0.9961869716644287, "reward_total_composite_std": 0.0016419171588495374, "reward_total_mean": 0.9961869716644287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961869716644287, "rewards/meter/std": 0.0016419171588495374, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961869716644287, "rewards/total_composite/std": 0.0016419171588495374, "sampling/importance_sampling_ratio/max": 1.939770221710205, "sampling/importance_sampling_ratio/mean": 1.0081746578216553, "sampling/importance_sampling_ratio/min": 0.4317508041858673, "sampling/sampling_logp_difference/max": 0.8399066925048828, "sampling/sampling_logp_difference/mean": 0.036743052303791046, "step": 1640 }, { "clip_ratio/high_max": 0.047490152064710855, "clip_ratio/high_mean": 0.047490152064710855, "clip_ratio/low_mean": 0.01592771988362074, "clip_ratio/low_min": 0.01592771988362074, "clip_ratio/region_mean": 0.0634178719483316, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 69.375, "completions/mean_terminated_length": 69.375, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.9592136070132256, "epoch": 0.06591155560910954, "frac_reward_zero_std": 0.0, "grad_norm": 6.707535266876221, "learning_rate": 5.030303030303031e-06, "loss": -0.0356, "num_tokens": 3690011.0, "reward": 0.7832512855529785, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7832512855529785, "reward_meter_std": 0.33663251996040344, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.33663249015808105, "reward_total_composite_mean": 0.7832512855529785, "reward_total_composite_std": 0.33663251996040344, "reward_total_mean": 0.7832512855529785, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7832512855529785, "rewards/meter/std": 0.33663251996040344, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7832512855529785, "rewards/total_composite/std": 0.33663251996040344, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0248838663101196, "sampling/importance_sampling_ratio/min": 0.35646137595176697, "sampling/sampling_logp_difference/max": 1.031529426574707, "sampling/sampling_logp_difference/mean": 0.09360064566135406, "step": 1641 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.0062332607340067625, "epoch": 0.06595172109089449, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.027272727272728e-06, "loss": 0.0, "num_tokens": 3691659.0, "reward": 0.993520200252533, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.993520200252533, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.993520200252533, "reward_total_composite_std": 0.0, "reward_total_mean": 0.993520200252533, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.993520200252533, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993520200252533, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0146211385726929, "sampling/importance_sampling_ratio/mean": 1.0005112886428833, "sampling/importance_sampling_ratio/min": 0.9973773956298828, "sampling/sampling_logp_difference/max": 0.014515344053506851, "sampling/sampling_logp_difference/mean": 0.0005443372065201402, "step": 1642 }, { "clip_ratio/high_max": 0.042821857146918774, "clip_ratio/high_mean": 0.042821857146918774, "clip_ratio/low_mean": 0.009563126135617495, "clip_ratio/low_min": 0.009563126135617495, "clip_ratio/region_mean": 0.05238498328253627, "completions/clipped_ratio": 0.0, "completions/max_length": 227.0, "completions/max_terminated_length": 227.0, "completions/mean_length": 203.5, "completions/mean_terminated_length": 203.5, "completions/min_length": 193.0, "completions/min_terminated_length": 193.0, "entropy": 0.8424399569630623, "epoch": 0.06599188657267945, "frac_reward_zero_std": 0.0, "grad_norm": 3.824038028717041, "learning_rate": 5.024242424242425e-06, "loss": 0.0248, "num_tokens": 3695039.0, "reward": 0.9925307035446167, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9925307035446167, "reward_meter_std": 0.010234549641609192, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010234540328383446, "reward_total_composite_mean": 0.9925307035446167, "reward_total_composite_std": 0.010234549641609192, "reward_total_mean": 0.9925307035446167, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9925307035446167, "rewards/meter/std": 0.010234549641609192, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9925307035446167, "rewards/total_composite/std": 0.010234549641609192, "sampling/importance_sampling_ratio/max": 1.9295165538787842, "sampling/importance_sampling_ratio/mean": 1.0239778757095337, "sampling/importance_sampling_ratio/min": 0.29857030510902405, "sampling/sampling_logp_difference/max": 1.208749771118164, "sampling/sampling_logp_difference/mean": 0.07599979639053345, "step": 1643 }, { "clip_ratio/high_max": 0.037598957773298025, "clip_ratio/high_mean": 0.037598957773298025, "clip_ratio/low_mean": 0.016500307247042656, "clip_ratio/low_min": 0.016500307247042656, "clip_ratio/region_mean": 0.05409926502034068, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 146.875, "completions/mean_terminated_length": 146.875, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.6761485747992992, "epoch": 0.0660320520544644, "frac_reward_zero_std": 0.0, "grad_norm": 3.962242603302002, "learning_rate": 5.021212121212121e-06, "loss": -0.0099, "num_tokens": 3697742.0, "reward": 0.9379702806472778, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9914488792419434, "reward_meter_std": 0.019252995029091835, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07060354202985764, "reward_total_composite_mean": 0.9379702806472778, "reward_total_composite_std": 0.07060353457927704, "reward_total_mean": 0.9379702806472778, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9914488792419434, "rewards/meter/std": 0.019252995029091835, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9379702806472778, "rewards/total_composite/std": 0.07060353457927704, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0191352367401123, "sampling/importance_sampling_ratio/min": 0.24873672425746918, "sampling/sampling_logp_difference/max": 1.3913602828979492, "sampling/sampling_logp_difference/mean": 0.0685059130191803, "step": 1644 }, { "clip_ratio/high_max": 0.053518022410571575, "clip_ratio/high_mean": 0.053518022410571575, "clip_ratio/low_mean": 0.022980431327596307, "clip_ratio/low_min": 0.022980431327596307, "clip_ratio/region_mean": 0.07649845373816788, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 69.625, "completions/mean_terminated_length": 69.625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 1.080611765384674, "epoch": 0.06607221753624935, "frac_reward_zero_std": 0.0, "grad_norm": 7.402265548706055, "learning_rate": 5.0181818181818186e-06, "loss": -0.0102, "num_tokens": 3699539.0, "reward": 0.6198569536209106, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6198569536209106, "reward_meter_std": 0.3495922088623047, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3495922088623047, "reward_total_composite_mean": 0.6198569536209106, "reward_total_composite_std": 0.3495922088623047, "reward_total_mean": 0.6198569536209106, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6198569536209106, "rewards/meter/std": 0.3495922088623047, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6198569536209106, "rewards/total_composite/std": 0.3495922088623047, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0240029096603394, "sampling/importance_sampling_ratio/min": 0.1966705024242401, "sampling/sampling_logp_difference/max": 1.626225471496582, "sampling/sampling_logp_difference/mean": 0.10739893466234207, "step": 1645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.010349510470405221, "epoch": 0.06611238301803431, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.015151515151515e-06, "loss": 0.0, "num_tokens": 3701219.0, "reward": 0.993520200252533, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.993520200252533, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.993520200252533, "reward_total_composite_std": 0.0, "reward_total_mean": 0.993520200252533, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.993520200252533, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993520200252533, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0615943670272827, "sampling/importance_sampling_ratio/mean": 1.0011212825775146, "sampling/importance_sampling_ratio/min": 0.9319490194320679, "sampling/sampling_logp_difference/max": 0.07047711312770844, "sampling/sampling_logp_difference/mean": 0.0016902285860851407, "step": 1646 }, { "clip_ratio/high_max": 0.025925687979906797, "clip_ratio/high_mean": 0.025925687979906797, "clip_ratio/low_mean": 0.02626075316220522, "clip_ratio/low_min": 0.02626075316220522, "clip_ratio/region_mean": 0.05218644114211202, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 115.5, "completions/mean_terminated_length": 115.5, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.7444454878568649, "epoch": 0.06615254849981926, "frac_reward_zero_std": 0.0, "grad_norm": 4.599454879760742, "learning_rate": 5.012121212121212e-06, "loss": 0.0056, "num_tokens": 3703551.0, "reward": 0.989443302154541, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.989443302154541, "reward_meter_std": 0.009337217546999454, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009337219409644604, "reward_total_composite_mean": 0.989443302154541, "reward_total_composite_std": 0.009337217546999454, "reward_total_mean": 0.989443302154541, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.989443302154541, "rewards/meter/std": 0.009337217546999454, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.989443302154541, "rewards/total_composite/std": 0.009337217546999454, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0189144611358643, "sampling/importance_sampling_ratio/min": 0.2730981409549713, "sampling/sampling_logp_difference/max": 1.2979240417480469, "sampling/sampling_logp_difference/mean": 0.07534082978963852, "step": 1647 }, { "clip_ratio/high_max": 0.06016605906188488, "clip_ratio/high_mean": 0.06016605906188488, "clip_ratio/low_mean": 0.01907266629859805, "clip_ratio/low_min": 0.01907266629859805, "clip_ratio/region_mean": 0.07923872536048293, "completions/clipped_ratio": 0.0, "completions/max_length": 112.0, "completions/max_terminated_length": 112.0, "completions/mean_length": 103.375, "completions/mean_terminated_length": 103.375, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 1.183668315410614, "epoch": 0.06619271398160421, "frac_reward_zero_std": 0.0, "grad_norm": 5.8860039710998535, "learning_rate": 5.009090909090909e-06, "loss": 0.0453, "num_tokens": 3705642.0, "reward": 0.877294659614563, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.877294659614563, "reward_meter_std": 0.19824109971523285, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.19824109971523285, "reward_total_composite_mean": 0.877294659614563, "reward_total_composite_std": 0.19824109971523285, "reward_total_mean": 0.877294659614563, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.877294659614563, "rewards/meter/std": 0.19824109971523285, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.877294659614563, "rewards/total_composite/std": 0.19824109971523285, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.031799077987671, "sampling/importance_sampling_ratio/min": 0.1795956790447235, "sampling/sampling_logp_difference/max": 1.7170472145080566, "sampling/sampling_logp_difference/mean": 0.10711432248353958, "step": 1648 }, { "clip_ratio/high_max": 0.03531280532479286, "clip_ratio/high_mean": 0.03531280532479286, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/region_mean": 0.04337732121348381, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.4391027558594942, "epoch": 0.06623287946338917, "frac_reward_zero_std": 0.0, "grad_norm": 5.83959436416626, "learning_rate": 5.006060606060607e-06, "loss": -0.0025, "num_tokens": 3707194.0, "reward": 0.9442778825759888, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9442778825759888, "reward_meter_std": 0.059931062161922455, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.059931062161922455, "reward_total_composite_mean": 0.9442778825759888, "reward_total_composite_std": 0.059931062161922455, "reward_total_mean": 0.9442778825759888, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9442778825759888, "rewards/meter/std": 0.059931062161922455, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9442778825759888, "rewards/total_composite/std": 0.059931062161922455, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007331371307373, "sampling/importance_sampling_ratio/min": 0.14738371968269348, "sampling/sampling_logp_difference/max": 1.9147157669067383, "sampling/sampling_logp_difference/mean": 0.05596788227558136, "step": 1649 }, { "clip_ratio/high_max": 0.0337981628254056, "clip_ratio/high_mean": 0.0337981628254056, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/region_mean": 0.03783042076975107, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 32.125, "completions/mean_terminated_length": 32.125, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.44733541272580624, "epoch": 0.06627304494517412, "frac_reward_zero_std": 0.0, "grad_norm": 4.691352367401123, "learning_rate": 5.003030303030303e-06, "loss": -0.0102, "num_tokens": 3708819.0, "reward": 0.9763838648796082, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9763838648796082, "reward_meter_std": 0.04593488201498985, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04593489319086075, "reward_total_composite_mean": 0.9763838648796082, "reward_total_composite_std": 0.04593488201498985, "reward_total_mean": 0.9763838648796082, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9763838648796082, "rewards/meter/std": 0.04593488201498985, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9763838648796082, "rewards/total_composite/std": 0.04593488201498985, "sampling/importance_sampling_ratio/max": 1.4046815633773804, "sampling/importance_sampling_ratio/mean": 1.0103769302368164, "sampling/importance_sampling_ratio/min": 0.2662525773048401, "sampling/sampling_logp_difference/max": 1.3233098983764648, "sampling/sampling_logp_difference/mean": 0.05742637813091278, "step": 1650 }, { "epoch": 0.06627304494517412, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/max_length": 409.3076923076923, "eval_completions/max_terminated_length": 372.9230769230769, "eval_completions/mean_length": 212.91346153846155, "eval_completions/mean_terminated_length": 204.67307927058295, "eval_completions/min_length": 62.0, "eval_completions/min_terminated_length": 62.0, "eval_entropy": 0.4149218121400246, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 3708819.0, "eval_reward": 0.5518424786054171, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.9108829039793748, "eval_reward_count_adherence_std": 0.11557848694232795, "eval_reward_meter_mean": 0.7244124962733343, "eval_reward_meter_std": 0.36815990120745623, "eval_reward_repeat_penalty_mean": 0.8433380264502305, "eval_reward_repeat_penalty_std": 0.15947996452450752, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5518424786054171, "eval_reward_total_composite_std": 0.34955332485529095, "eval_reward_total_mean": 0.5518424786054171, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.9108829039793748, "eval_rewards/count_adherence/std": 0.11557848694232795, "eval_rewards/meter/mean": 0.7244124962733343, "eval_rewards/meter/std": 0.36815990120745623, "eval_rewards/repeat_penalty/mean": 0.8433380264502305, "eval_rewards/repeat_penalty/std": 0.15947996452450752, "eval_rewards/total_composite/mean": 0.5518424786054171, "eval_rewards/total_composite/std": 0.34955332485529095, "eval_runtime": 79.2708, "eval_samples_per_second": 1.312, "eval_sampling/importance_sampling_ratio/max": 1.5815464166494517, "eval_sampling/importance_sampling_ratio/mean": 1.009941733800448, "eval_sampling/importance_sampling_ratio/min": 0.3488806807077848, "eval_sampling/sampling_logp_difference/max": 1.0809908371705275, "eval_sampling/sampling_logp_difference/mean": 0.035442069316139586, "eval_steps_per_second": 0.164, "step": 1650 }, { "clip_ratio/high_max": 0.005569578497670591, "clip_ratio/high_mean": 0.005569578497670591, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.007407813798636198, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.625, "completions/mean_terminated_length": 67.625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.06585087720304728, "epoch": 0.06631321042695908, "frac_reward_zero_std": 0.0, "grad_norm": 0.5019415616989136, "learning_rate": 5e-06, "loss": 0.0001, "num_tokens": 3710592.0, "reward": 0.9974769949913025, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974769949913025, "reward_meter_std": 0.001003190758638084, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001003190758638084, "reward_total_composite_mean": 0.9974769949913025, "reward_total_composite_std": 0.001003190758638084, "reward_total_mean": 0.9974769949913025, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974769949913025, "rewards/meter/std": 0.001003190758638084, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974769949913025, "rewards/total_composite/std": 0.001003190758638084, "sampling/importance_sampling_ratio/max": 1.4636121988296509, "sampling/importance_sampling_ratio/mean": 1.0023994445800781, "sampling/importance_sampling_ratio/min": 0.5320457816123962, "sampling/sampling_logp_difference/max": 0.6310257911682129, "sampling/sampling_logp_difference/mean": 0.009622362442314625, "step": 1651 }, { "clip_ratio/high_max": 0.007299659075215459, "clip_ratio/high_mean": 0.007299659075215459, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007299659075215459, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.875, "completions/mean_terminated_length": 67.875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.07074726792052388, "epoch": 0.06635337590874403, "frac_reward_zero_std": 0.0, "grad_norm": 1.1626874208450317, "learning_rate": 4.996969696969698e-06, "loss": -0.0038, "num_tokens": 3712383.0, "reward": 0.9978716373443604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978716373443604, "reward_meter_std": 7.145507697714493e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.145912968553603e-05, "reward_total_composite_mean": 0.9978716373443604, "reward_total_composite_std": 7.145507697714493e-05, "reward_total_mean": 0.9978716373443604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978716373443604, "rewards/meter/std": 7.145507697714493e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978716373443604, "rewards/total_composite/std": 7.145507697714493e-05, "sampling/importance_sampling_ratio/max": 1.2595540285110474, "sampling/importance_sampling_ratio/mean": 1.0000650882720947, "sampling/importance_sampling_ratio/min": 0.4046737849712372, "sampling/sampling_logp_difference/max": 0.9046740531921387, "sampling/sampling_logp_difference/mean": 0.011869369074702263, "step": 1652 }, { "clip_ratio/high_max": 0.03797963773831725, "clip_ratio/high_mean": 0.03797963773831725, "clip_ratio/low_mean": 0.007788162212818861, "clip_ratio/low_min": 0.007788162212818861, "clip_ratio/region_mean": 0.04576779995113611, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 382.0, "completions/mean_terminated_length": 338.66668701171875, "completions/min_length": 321.0, "completions/min_terminated_length": 321.0, "entropy": 0.7133785486221313, "epoch": 0.06639354139052898, "frac_reward_zero_std": 0.0, "grad_norm": 1.3737280368804932, "learning_rate": 4.993939393939394e-06, "loss": -0.3504, "num_tokens": 3716159.0, "reward": 0.5758238434791565, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8152369260787964, "reward_meter_std": 0.1942194551229477, "reward_repeat_penalty_mean": 0.9428104758262634, "reward_repeat_penalty_std": 0.05349903926253319, "reward_std": 0.36915671825408936, "reward_total_composite_mean": 0.5758238434791565, "reward_total_composite_std": 0.36915671825408936, "reward_total_mean": 0.5758238434791565, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8152369260787964, "rewards/meter/std": 0.1942194551229477, "rewards/repeat_penalty/mean": 0.9428104758262634, "rewards/repeat_penalty/std": 0.05349903926253319, "rewards/total_composite/mean": 0.5758238434791565, "rewards/total_composite/std": 0.36915671825408936, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0216100215911865, "sampling/importance_sampling_ratio/min": 0.11343610286712646, "sampling/sampling_logp_difference/max": 2.176515579223633, "sampling/sampling_logp_difference/mean": 0.08613347262144089, "step": 1653 }, { "clip_ratio/high_max": 0.00562528264708817, "clip_ratio/high_mean": 0.00562528264708817, "clip_ratio/low_mean": 0.010080644860863686, "clip_ratio/low_min": 0.010080644860863686, "clip_ratio/region_mean": 0.015705927507951856, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 63.625, "completions/mean_terminated_length": 63.625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.0451802066527307, "epoch": 0.06643370687231394, "frac_reward_zero_std": 0.0, "grad_norm": 3.5595715045928955, "learning_rate": 4.990909090909091e-06, "loss": -0.0232, "num_tokens": 3718004.0, "reward": 0.9683968424797058, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9683968424797058, "reward_meter_std": 0.024647340178489685, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.024647342041134834, "reward_total_composite_mean": 0.9683968424797058, "reward_total_composite_std": 0.024647340178489685, "reward_total_mean": 0.9683968424797058, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9683968424797058, "rewards/meter/std": 0.024647340178489685, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9683968424797058, "rewards/total_composite/std": 0.024647340178489685, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9969826936721802, "sampling/importance_sampling_ratio/min": 0.00629988731816411, "sampling/sampling_logp_difference/max": 5.06722354888916, "sampling/sampling_logp_difference/mean": 0.03070976585149765, "step": 1654 }, { "clip_ratio/high_max": 0.014928699005395174, "clip_ratio/high_mean": 0.014928699005395174, "clip_ratio/low_mean": 0.0078125, "clip_ratio/low_min": 0.0078125, "clip_ratio/region_mean": 0.022741199005395174, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.25, "completions/mean_terminated_length": 33.25, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.03417817200534046, "epoch": 0.06647387235409889, "frac_reward_zero_std": 0.0, "grad_norm": 5.8942670822143555, "learning_rate": 4.987878787878789e-06, "loss": -0.0142, "num_tokens": 3719438.0, "reward": 0.9977890849113464, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977890849113464, "reward_meter_std": 0.002695126226171851, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002695116912946105, "reward_total_composite_mean": 0.9977890849113464, "reward_total_composite_std": 0.002695126226171851, "reward_total_mean": 0.9977890849113464, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977890849113464, "rewards/meter/std": 0.002695126226171851, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977890849113464, "rewards/total_composite/std": 0.002695126226171851, "sampling/importance_sampling_ratio/max": 1.4004013538360596, "sampling/importance_sampling_ratio/mean": 0.9885497093200684, "sampling/importance_sampling_ratio/min": 0.016080791130661964, "sampling/sampling_logp_difference/max": 4.130129814147949, "sampling/sampling_logp_difference/mean": 0.0360952690243721, "step": 1655 }, { "clip_ratio/high_max": 0.046408120542764664, "clip_ratio/high_mean": 0.046408120542764664, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.046408120542764664, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 214.0, "completions/mean_length": 235.0, "completions/mean_terminated_length": 195.42857360839844, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "entropy": 0.6871326193213463, "epoch": 0.06651403783588385, "frac_reward_zero_std": 0.0, "grad_norm": 1.0378773212432861, "learning_rate": 4.984848484848485e-06, "loss": -0.2605, "num_tokens": 3722366.0, "reward": 0.8588160872459412, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.925000011920929, "reward_count_adherence_std": 0.2121320217847824, "reward_meter_mean": 0.8787662982940674, "reward_meter_std": 0.3354223370552063, "reward_repeat_penalty_mean": 0.9861111044883728, "reward_repeat_penalty_std": 0.03928370773792267, "reward_std": 0.34913378953933716, "reward_total_composite_mean": 0.8588160872459412, "reward_total_composite_std": 0.34913378953933716, "reward_total_mean": 0.8588160872459412, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.925000011920929, "rewards/count_adherence/std": 0.2121320217847824, "rewards/meter/mean": 0.8787662982940674, "rewards/meter/std": 0.3354223370552063, "rewards/repeat_penalty/mean": 0.9861111044883728, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.8588160872459412, "rewards/total_composite/std": 0.34913378953933716, "sampling/importance_sampling_ratio/max": 1.8245017528533936, "sampling/importance_sampling_ratio/mean": 1.0213563442230225, "sampling/importance_sampling_ratio/min": 0.17044617235660553, "sampling/sampling_logp_difference/max": 1.7693357467651367, "sampling/sampling_logp_difference/mean": 0.0746423751115799, "step": 1656 }, { "clip_ratio/high_max": 0.021861399058252573, "clip_ratio/high_mean": 0.021861399058252573, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.021861399058252573, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 63.25, "completions/mean_terminated_length": 63.25, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.3440733216702938, "epoch": 0.0665542033176688, "frac_reward_zero_std": 0.0, "grad_norm": 6.6080169677734375, "learning_rate": 4.981818181818182e-06, "loss": -0.0037, "num_tokens": 3724136.0, "reward": 0.9078664779663086, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9078664779663086, "reward_meter_std": 0.15767432749271393, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.15767431259155273, "reward_total_composite_mean": 0.9078664779663086, "reward_total_composite_std": 0.15767432749271393, "reward_total_mean": 0.9078664779663086, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9078664779663086, "rewards/meter/std": 0.15767432749271393, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9078664779663086, "rewards/total_composite/std": 0.15767432749271393, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.998699963092804, "sampling/importance_sampling_ratio/min": 0.14266319572925568, "sampling/sampling_logp_difference/max": 1.9472687244415283, "sampling/sampling_logp_difference/mean": 0.043513912707567215, "step": 1657 }, { "clip_ratio/high_max": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/region_mean": 0.003876201924867928, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.5, "completions/mean_terminated_length": 64.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.02129516121931374, "epoch": 0.06659436879945375, "frac_reward_zero_std": 0.0, "grad_norm": 2.866943597793579, "learning_rate": 4.978787878787879e-06, "loss": 0.0035, "num_tokens": 3726020.0, "reward": 0.9981748461723328, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981748461723328, "reward_meter_std": 7.74198560975492e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.74198560975492e-05, "reward_total_composite_mean": 0.9981748461723328, "reward_total_composite_std": 7.74198560975492e-05, "reward_total_mean": 0.9981748461723328, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981748461723328, "rewards/meter/std": 7.74198560975492e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981748461723328, "rewards/total_composite/std": 7.74198560975492e-05, "sampling/importance_sampling_ratio/max": 1.3977038860321045, "sampling/importance_sampling_ratio/mean": 0.9998639822006226, "sampling/importance_sampling_ratio/min": 0.22167760133743286, "sampling/sampling_logp_difference/max": 1.5065312385559082, "sampling/sampling_logp_difference/mean": 0.007607936859130859, "step": 1658 }, { "clip_ratio/high_max": 0.005542142200283706, "clip_ratio/high_mean": 0.005542142200283706, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005542142200283706, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.06244664313271642, "epoch": 0.06663453428123871, "frac_reward_zero_std": 0.0, "grad_norm": 5.0424885749816895, "learning_rate": 4.975757575757576e-06, "loss": 0.0129, "num_tokens": 3727828.0, "reward": 0.9952208995819092, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952208995819092, "reward_meter_std": 0.004935278091579676, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004935278091579676, "reward_total_composite_mean": 0.9952208995819092, "reward_total_composite_std": 0.004935278091579676, "reward_total_mean": 0.9952208995819092, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952208995819092, "rewards/meter/std": 0.004935278091579676, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952208995819092, "rewards/total_composite/std": 0.004935278091579676, "sampling/importance_sampling_ratio/max": 1.3186924457550049, "sampling/importance_sampling_ratio/mean": 1.0028293132781982, "sampling/importance_sampling_ratio/min": 0.42393937706947327, "sampling/sampling_logp_difference/max": 0.8581647872924805, "sampling/sampling_logp_difference/mean": 0.009829939343035221, "step": 1659 }, { "clip_ratio/high_max": 0.0017985611921176314, "clip_ratio/high_mean": 0.0017985611921176314, "clip_ratio/low_mean": 0.0008992805960588157, "clip_ratio/low_min": 0.0008992805960588157, "clip_ratio/region_mean": 0.002697841788176447, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 139.125, "completions/mean_terminated_length": 139.125, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.016455981065519154, "epoch": 0.06667469976302366, "frac_reward_zero_std": 0.0, "grad_norm": 0.4592190384864807, "learning_rate": 4.972727272727273e-06, "loss": -0.0015, "num_tokens": 3730317.0, "reward": 0.7461007833480835, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948010444641113, "reward_meter_std": 0.004489070735871792, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0033668053802102804, "reward_total_composite_mean": 0.7461007833480835, "reward_total_composite_std": 0.003366807708516717, "reward_total_mean": 0.7461007833480835, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948010444641113, "rewards/meter/std": 0.004489070735871792, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7461007833480835, "rewards/total_composite/std": 0.003366807708516717, "sampling/importance_sampling_ratio/max": 1.1272794008255005, "sampling/importance_sampling_ratio/mean": 0.9996078014373779, "sampling/importance_sampling_ratio/min": 0.34133151173591614, "sampling/sampling_logp_difference/max": 1.0749011039733887, "sampling/sampling_logp_difference/mean": 0.004387721884995699, "step": 1660 }, { "clip_ratio/high_max": 0.008511104620993137, "clip_ratio/high_mean": 0.008511104620993137, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.008511104620993137, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 58.875, "completions/mean_terminated_length": 58.875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.14373003505170345, "epoch": 0.06671486524480862, "frac_reward_zero_std": 0.0, "grad_norm": 5.378631114959717, "learning_rate": 4.9696969696969696e-06, "loss": 0.0154, "num_tokens": 3732140.0, "reward": 0.9912177324295044, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9912177324295044, "reward_meter_std": 0.010196500457823277, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010196508839726448, "reward_total_composite_mean": 0.9912177324295044, "reward_total_composite_std": 0.010196500457823277, "reward_total_mean": 0.9912177324295044, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9912177324295044, "rewards/meter/std": 0.010196500457823277, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9912177324295044, "rewards/total_composite/std": 0.010196500457823277, "sampling/importance_sampling_ratio/max": 1.3040835857391357, "sampling/importance_sampling_ratio/mean": 1.0054762363433838, "sampling/importance_sampling_ratio/min": 0.5027512907981873, "sampling/sampling_logp_difference/max": 0.687659740447998, "sampling/sampling_logp_difference/mean": 0.015276388265192509, "step": 1661 }, { "clip_ratio/high_max": 0.008474576286971569, "clip_ratio/high_mean": 0.008474576286971569, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/region_mean": 0.016539092175662518, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 59.25, "completions/mean_terminated_length": 59.25, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.1442255014553666, "epoch": 0.06675503072659357, "frac_reward_zero_std": 0.0, "grad_norm": 3.579711675643921, "learning_rate": 4.966666666666667e-06, "loss": 0.0131, "num_tokens": 3733822.0, "reward": 0.9940330982208252, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9940330982208252, "reward_meter_std": 0.002611657604575157, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002611662959679961, "reward_total_composite_mean": 0.9940330982208252, "reward_total_composite_std": 0.002611657604575157, "reward_total_mean": 0.9940330982208252, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9940330982208252, "rewards/meter/std": 0.002611657604575157, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9940330982208252, "rewards/total_composite/std": 0.002611657604575157, "sampling/importance_sampling_ratio/max": 1.5864135026931763, "sampling/importance_sampling_ratio/mean": 1.0035570859909058, "sampling/importance_sampling_ratio/min": 0.4705776870250702, "sampling/sampling_logp_difference/max": 0.7537941932678223, "sampling/sampling_logp_difference/mean": 0.015829751268029213, "step": 1662 }, { "clip_ratio/high_max": 0.03229976072907448, "clip_ratio/high_mean": 0.03229976072907448, "clip_ratio/low_mean": 0.010420631850138307, "clip_ratio/low_min": 0.010420631850138307, "clip_ratio/region_mean": 0.042720392579212785, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 72.75, "completions/mean_terminated_length": 72.75, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.49573158100247383, "epoch": 0.06679519620837852, "frac_reward_zero_std": 0.0, "grad_norm": 4.979900360107422, "learning_rate": 4.963636363636364e-06, "loss": 0.0149, "num_tokens": 3735532.0, "reward": 0.9954915046691895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9954915046691895, "reward_meter_std": 0.002510088263079524, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002510099671781063, "reward_total_composite_mean": 0.9954915046691895, "reward_total_composite_std": 0.002510088263079524, "reward_total_mean": 0.9954915046691895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9954915046691895, "rewards/meter/std": 0.002510088263079524, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9954915046691895, "rewards/total_composite/std": 0.002510088263079524, "sampling/importance_sampling_ratio/max": 1.7458137273788452, "sampling/importance_sampling_ratio/mean": 1.0033316612243652, "sampling/importance_sampling_ratio/min": 0.31032371520996094, "sampling/sampling_logp_difference/max": 1.1701393127441406, "sampling/sampling_logp_difference/mean": 0.06602619588375092, "step": 1663 }, { "clip_ratio/high_max": 0.028268360998481512, "clip_ratio/high_mean": 0.028268360998481512, "clip_ratio/low_mean": 0.004454409005120397, "clip_ratio/low_min": 0.004454409005120397, "clip_ratio/region_mean": 0.03272277000360191, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 109.75, "completions/mean_terminated_length": 109.75, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.3692820519208908, "epoch": 0.06683536169016348, "frac_reward_zero_std": 0.0, "grad_norm": 3.2529075145721436, "learning_rate": 4.9606060606060605e-06, "loss": -0.0026, "num_tokens": 3737762.0, "reward": 0.9983501434326172, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983501434326172, "reward_meter_std": 0.000316560355713591, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0003165603266097605, "reward_total_composite_mean": 0.9983501434326172, "reward_total_composite_std": 0.000316560355713591, "reward_total_mean": 0.9983501434326172, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983501434326172, "rewards/meter/std": 0.000316560355713591, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983501434326172, "rewards/total_composite/std": 0.000316560355713591, "sampling/importance_sampling_ratio/max": 1.7752270698547363, "sampling/importance_sampling_ratio/mean": 1.0035275220870972, "sampling/importance_sampling_ratio/min": 0.32426539063453674, "sampling/sampling_logp_difference/max": 1.1261930465698242, "sampling/sampling_logp_difference/mean": 0.05091632902622223, "step": 1664 }, { "clip_ratio/high_max": 0.029565565433586016, "clip_ratio/high_mean": 0.029565565433586016, "clip_ratio/low_mean": 0.004957506898790598, "clip_ratio/low_min": 0.004957506898790598, "clip_ratio/region_mean": 0.034523072332376614, "completions/clipped_ratio": 0.0, "completions/max_length": 366.0, "completions/max_terminated_length": 366.0, "completions/mean_length": 340.625, "completions/mean_terminated_length": 340.625, "completions/min_length": 320.0, "completions/min_terminated_length": 320.0, "entropy": 0.3913237787783146, "epoch": 0.06687552717194843, "frac_reward_zero_std": 0.0, "grad_norm": 2.9181759357452393, "learning_rate": 4.957575757575758e-06, "loss": 0.0217, "num_tokens": 3742119.0, "reward": 0.5716930627822876, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9840777516365051, "reward_meter_std": 0.037384092807769775, "reward_repeat_penalty_mean": 0.7647058963775635, "reward_repeat_penalty_std": 0.10428297519683838, "reward_std": 0.2410053312778473, "reward_total_composite_mean": 0.5716930627822876, "reward_total_composite_std": 0.2410053312778473, "reward_total_mean": 0.5716930627822876, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9840777516365051, "rewards/meter/std": 0.037384092807769775, "rewards/repeat_penalty/mean": 0.7647058963775635, "rewards/repeat_penalty/std": 0.10428297519683838, "rewards/total_composite/mean": 0.5716930627822876, "rewards/total_composite/std": 0.2410053312778473, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0077635049819946, "sampling/importance_sampling_ratio/min": 3.980578185291961e-05, "sampling/sampling_logp_difference/max": 10.131498336791992, "sampling/sampling_logp_difference/mean": 0.05633535981178284, "step": 1665 }, { "clip_ratio/high_max": 0.01824606256559491, "clip_ratio/high_mean": 0.01824606256559491, "clip_ratio/low_mean": 0.01092962856637314, "clip_ratio/low_min": 0.01092962856637314, "clip_ratio/region_mean": 0.02917569113196805, "completions/clipped_ratio": 0.0, "completions/max_length": 314.0, "completions/max_terminated_length": 314.0, "completions/mean_length": 302.125, "completions/mean_terminated_length": 302.125, "completions/min_length": 285.0, "completions/min_terminated_length": 285.0, "entropy": 0.2841501086950302, "epoch": 0.06691569265373339, "frac_reward_zero_std": 0.0, "grad_norm": 1.908144474029541, "learning_rate": 4.954545454545455e-06, "loss": -0.0177, "num_tokens": 3746080.0, "reward": 0.5741910934448242, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8181818127632141, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9341138005256653, "reward_meter_std": 0.05671977251768112, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.08169000595808029, "reward_std": 0.07986099272966385, "reward_total_composite_mean": 0.5741910934448242, "reward_total_composite_std": 0.07986098527908325, "reward_total_mean": 0.5741910934448242, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8181818127632141, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9341138005256653, "rewards/meter/std": 0.05671977251768112, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.08169000595808029, "rewards/total_composite/mean": 0.5741910934448242, "rewards/total_composite/std": 0.07986098527908325, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0056406259536743, "sampling/importance_sampling_ratio/min": 0.08349069207906723, "sampling/sampling_logp_difference/max": 2.483020067214966, "sampling/sampling_logp_difference/mean": 0.03923000767827034, "step": 1666 }, { "clip_ratio/high_max": 0.004938361467793584, "clip_ratio/high_mean": 0.004938361467793584, "clip_ratio/low_mean": 0.0024390824837610126, "clip_ratio/low_min": 0.0024390824837610126, "clip_ratio/region_mean": 0.007377443951554596, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 101.75, "completions/mean_terminated_length": 101.75, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.08447205042466521, "epoch": 0.06695585813551834, "frac_reward_zero_std": 0.0, "grad_norm": 2.4326391220092773, "learning_rate": 4.951515151515152e-06, "loss": 0.0056, "num_tokens": 3748310.0, "reward": 0.7976030111312866, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970037937164307, "reward_meter_std": 0.0013122691307216883, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010498074116185308, "reward_total_composite_mean": 0.7976030111312866, "reward_total_composite_std": 0.0010498042684048414, "reward_total_mean": 0.7976030111312866, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970037937164307, "rewards/meter/std": 0.0013122691307216883, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7976030111312866, "rewards/total_composite/std": 0.0010498042684048414, "sampling/importance_sampling_ratio/max": 1.46913480758667, "sampling/importance_sampling_ratio/mean": 1.0012123584747314, "sampling/importance_sampling_ratio/min": 0.33690041303634644, "sampling/sampling_logp_difference/max": 1.087967872619629, "sampling/sampling_logp_difference/mean": 0.010477752424776554, "step": 1667 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 133.75, "completions/mean_terminated_length": 133.75, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.027194248279556632, "epoch": 0.0669960236173033, "frac_reward_zero_std": 0.0, "grad_norm": 0.08446517586708069, "learning_rate": 4.9484848484848495e-06, "loss": -0.0005, "num_tokens": 3750844.0, "reward": 0.7128248810768127, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979548454284668, "reward_meter_std": 2.2615500711253844e-05, "reward_repeat_penalty_mean": 0.7142857313156128, "reward_repeat_penalty_std": 0.0, "reward_std": 1.6179135855054483e-05, "reward_total_composite_mean": 0.7128248810768127, "reward_total_composite_std": 1.6168985894182697e-05, "reward_total_mean": 0.7128248810768127, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979548454284668, "rewards/meter/std": 2.2615500711253844e-05, "rewards/repeat_penalty/mean": 0.7142857313156128, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7128248810768127, "rewards/total_composite/std": 1.6168985894182697e-05, "sampling/importance_sampling_ratio/max": 1.2750511169433594, "sampling/importance_sampling_ratio/mean": 1.001842975616455, "sampling/importance_sampling_ratio/min": 0.36468705534935, "sampling/sampling_logp_difference/max": 1.0087156295776367, "sampling/sampling_logp_difference/mean": 0.004637876525521278, "step": 1668 }, { "clip_ratio/high_max": 0.0018115942366421223, "clip_ratio/high_mean": 0.0018115942366421223, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018115942366421223, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.03746737586334348, "epoch": 0.06703618909908825, "frac_reward_zero_std": 0.0, "grad_norm": 0.0685642808675766, "learning_rate": 4.945454545454546e-06, "loss": -0.0009, "num_tokens": 3752596.0, "reward": 0.9978952407836914, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978952407836914, "reward_meter_std": 1.1280608305241913e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.128060739574721e-05, "reward_total_composite_mean": 0.9978952407836914, "reward_total_composite_std": 1.1280608305241913e-05, "reward_total_mean": 0.9978952407836914, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978952407836914, "rewards/meter/std": 1.1280608305241913e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978952407836914, "rewards/total_composite/std": 1.1280608305241913e-05, "sampling/importance_sampling_ratio/max": 1.0789562463760376, "sampling/importance_sampling_ratio/mean": 1.0022841691970825, "sampling/importance_sampling_ratio/min": 0.6614209413528442, "sampling/sampling_logp_difference/max": 0.41336488723754883, "sampling/sampling_logp_difference/mean": 0.004578685853630304, "step": 1669 }, { "clip_ratio/high_max": 0.01764108322095126, "clip_ratio/high_mean": 0.01764108322095126, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/region_mean": 0.021673341165296733, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 63.625, "completions/mean_terminated_length": 63.625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.1983039677143097, "epoch": 0.0670763545808732, "frac_reward_zero_std": 0.0, "grad_norm": 3.68241024017334, "learning_rate": 4.942424242424243e-06, "loss": 0.0003, "num_tokens": 3754401.0, "reward": 0.986209511756897, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.986209511756897, "reward_meter_std": 0.004416565876454115, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004416569136083126, "reward_total_composite_mean": 0.986209511756897, "reward_total_composite_std": 0.004416565876454115, "reward_total_mean": 0.986209511756897, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.986209511756897, "rewards/meter/std": 0.004416565876454115, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.986209511756897, "rewards/total_composite/std": 0.004416565876454115, "sampling/importance_sampling_ratio/max": 1.363709807395935, "sampling/importance_sampling_ratio/mean": 1.000848650932312, "sampling/importance_sampling_ratio/min": 0.2110455483198166, "sampling/sampling_logp_difference/max": 1.5556812286376953, "sampling/sampling_logp_difference/mean": 0.029973594471812248, "step": 1670 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0012499999720603228, "clip_ratio/low_min": 0.0012499999720603228, "clip_ratio/region_mean": 0.0012499999720603228, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 100.75, "completions/mean_terminated_length": 100.75, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.022657749010249972, "epoch": 0.06711652006265816, "frac_reward_zero_std": 0.0, "grad_norm": 1.1145938634872437, "learning_rate": 4.93939393939394e-06, "loss": -0.0002, "num_tokens": 3756527.0, "reward": 0.8232920169830322, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979295134544373, "reward_meter_std": 2.0676665371865965e-05, "reward_repeat_penalty_mean": 0.8250000476837158, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07056733220815659, "reward_total_composite_mean": 0.8232920169830322, "reward_total_composite_std": 0.07056734710931778, "reward_total_mean": 0.8232920169830322, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979295134544373, "rewards/meter/std": 2.0676665371865965e-05, "rewards/repeat_penalty/mean": 0.8250000476837158, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8232920169830322, "rewards/total_composite/std": 0.07056734710931778, "sampling/importance_sampling_ratio/max": 1.0940825939178467, "sampling/importance_sampling_ratio/mean": 1.0012245178222656, "sampling/importance_sampling_ratio/min": 0.809956967830658, "sampling/sampling_logp_difference/max": 0.21077418327331543, "sampling/sampling_logp_difference/mean": 0.0027057162951678038, "step": 1671 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.875, "completions/mean_terminated_length": 32.875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.02497886586934328, "epoch": 0.06715668554444311, "frac_reward_zero_std": 0.0, "grad_norm": 4.178384304046631, "learning_rate": 4.936363636363637e-06, "loss": -0.0112, "num_tokens": 3758078.0, "reward": 0.9963486194610596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963486194610596, "reward_meter_std": 0.006714941468089819, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006714944262057543, "reward_total_composite_mean": 0.9963486194610596, "reward_total_composite_std": 0.006714941468089819, "reward_total_mean": 0.9963486194610596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963486194610596, "rewards/meter/std": 0.006714941468089819, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963486194610596, "rewards/total_composite/std": 0.006714941468089819, "sampling/importance_sampling_ratio/max": 1.1358792781829834, "sampling/importance_sampling_ratio/mean": 1.000542163848877, "sampling/importance_sampling_ratio/min": 0.6010414361953735, "sampling/sampling_logp_difference/max": 0.5090913772583008, "sampling/sampling_logp_difference/mean": 0.003993004094809294, "step": 1672 }, { "clip_ratio/high_max": 0.0351892322069034, "clip_ratio/high_mean": 0.0351892322069034, "clip_ratio/low_mean": 0.008515364723280072, "clip_ratio/low_min": 0.008515364723280072, "clip_ratio/region_mean": 0.04370459693018347, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.47178683429956436, "epoch": 0.06719685102622806, "frac_reward_zero_std": 0.0, "grad_norm": 5.359276294708252, "learning_rate": 4.933333333333334e-06, "loss": 0.0159, "num_tokens": 3760034.0, "reward": 0.995732307434082, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.995732307434082, "reward_meter_std": 0.00316053768619895, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0031605414114892483, "reward_total_composite_mean": 0.995732307434082, "reward_total_composite_std": 0.00316053768619895, "reward_total_mean": 0.995732307434082, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.995732307434082, "rewards/meter/std": 0.00316053768619895, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.995732307434082, "rewards/total_composite/std": 0.00316053768619895, "sampling/importance_sampling_ratio/max": 1.7035573720932007, "sampling/importance_sampling_ratio/mean": 1.0138006210327148, "sampling/importance_sampling_ratio/min": 0.2098330408334732, "sampling/sampling_logp_difference/max": 1.5614430904388428, "sampling/sampling_logp_difference/mean": 0.05547928065061569, "step": 1673 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.005282787780743092, "epoch": 0.06723701650801302, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.9303030303030305e-06, "loss": 0.0, "num_tokens": 3761666.0, "reward": 0.993520200252533, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.993520200252533, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.993520200252533, "reward_total_composite_std": 0.0, "reward_total_mean": 0.993520200252533, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.993520200252533, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993520200252533, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0186225175857544, "sampling/importance_sampling_ratio/mean": 1.0006948709487915, "sampling/importance_sampling_ratio/min": 0.9973061680793762, "sampling/sampling_logp_difference/max": 0.018451236188411713, "sampling/sampling_logp_difference/mean": 0.0007125565898604691, "step": 1674 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0014534883666783571, "clip_ratio/low_min": 0.0014534883666783571, "clip_ratio/region_mean": 0.0014534883666783571, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 86.0, "completions/mean_terminated_length": 86.0, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.017064974061213434, "epoch": 0.06727718198979797, "frac_reward_zero_std": 0.0, "grad_norm": 0.6250240802764893, "learning_rate": 4.927272727272728e-06, "loss": 0.0014, "num_tokens": 3763698.0, "reward": 0.7950762510299683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9938453435897827, "reward_meter_std": 0.0002583391033113003, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020667610806412995, "reward_total_composite_mean": 0.7950762510299683, "reward_total_composite_std": 0.00020666708587668836, "reward_total_mean": 0.7950762510299683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9938453435897827, "rewards/meter/std": 0.0002583391033113003, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7950762510299683, "rewards/total_composite/std": 0.00020666708587668836, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0030723810195923, "sampling/importance_sampling_ratio/min": 0.9702885746955872, "sampling/sampling_logp_difference/max": 1.0392670631408691, "sampling/sampling_logp_difference/mean": 0.0035054178442806005, "step": 1675 }, { "clip_ratio/high_max": 0.00492622796446085, "clip_ratio/high_mean": 0.00492622796446085, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.00492622796446085, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 101.0, "completions/mean_terminated_length": 101.0, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.04971472639590502, "epoch": 0.06731734747158293, "frac_reward_zero_std": 0.0, "grad_norm": 1.5943388938903809, "learning_rate": 4.924242424242425e-06, "loss": 0.0003, "num_tokens": 3765698.0, "reward": 0.8731323480606079, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978747963905334, "reward_meter_std": 0.0001633708452573046, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10320711135864258, "reward_total_composite_mean": 0.8731323480606079, "reward_total_composite_std": 0.10320712625980377, "reward_total_mean": 0.8731323480606079, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978747963905334, "rewards/meter/std": 0.0001633708452573046, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8731323480606079, "rewards/total_composite/std": 0.10320712625980377, "sampling/importance_sampling_ratio/max": 1.161500334739685, "sampling/importance_sampling_ratio/mean": 1.0001577138900757, "sampling/importance_sampling_ratio/min": 0.3650937080383301, "sampling/sampling_logp_difference/max": 1.007601261138916, "sampling/sampling_logp_difference/mean": 0.008153069764375687, "step": 1676 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.00934242527000606, "epoch": 0.06735751295336788, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.9212121212121214e-06, "loss": 0.0, "num_tokens": 3767450.0, "reward": 0.9982472658157349, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982472658157349, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9982472658157349, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9982472658157349, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982472658157349, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982472658157349, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0495675802230835, "sampling/importance_sampling_ratio/mean": 1.0002412796020508, "sampling/importance_sampling_ratio/min": 0.8145217299461365, "sampling/sampling_logp_difference/max": 0.2051541954278946, "sampling/sampling_logp_difference/mean": 0.0014719502069056034, "step": 1677 }, { "clip_ratio/high_max": 0.0037499999161809683, "clip_ratio/high_mean": 0.0037499999161809683, "clip_ratio/low_mean": 0.0012376237427815795, "clip_ratio/low_min": 0.0012376237427815795, "clip_ratio/region_mean": 0.004987623658962548, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 99.75, "completions/mean_terminated_length": 99.75, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.04197088209912181, "epoch": 0.06739767843515283, "frac_reward_zero_std": 0.0, "grad_norm": 1.9516417980194092, "learning_rate": 4.918181818181819e-06, "loss": 0.01, "num_tokens": 3769712.0, "reward": 0.748537540435791, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9527222514152527, "reward_meter_std": 0.0017279081512242556, "reward_repeat_penalty_mean": 0.7857142686843872, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.0724104791879654, "reward_total_composite_mean": 0.748537540435791, "reward_total_composite_std": 0.0724104791879654, "reward_total_mean": 0.748537540435791, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9527222514152527, "rewards/meter/std": 0.0017279081512242556, "rewards/repeat_penalty/mean": 0.7857142686843872, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.748537540435791, "rewards/total_composite/std": 0.0724104791879654, "sampling/importance_sampling_ratio/max": 1.2951877117156982, "sampling/importance_sampling_ratio/mean": 0.999224066734314, "sampling/importance_sampling_ratio/min": 0.19571292400360107, "sampling/sampling_logp_difference/max": 1.6311063766479492, "sampling/sampling_logp_difference/mean": 0.01071135699748993, "step": 1678 }, { "clip_ratio/high_max": 0.003399575361981988, "clip_ratio/high_mean": 0.003399575361981988, "clip_ratio/low_mean": 0.0011655074777081609, "clip_ratio/low_min": 0.0011655074777081609, "clip_ratio/region_mean": 0.004565082839690149, "completions/clipped_ratio": 0.0, "completions/max_length": 224.0, "completions/max_terminated_length": 224.0, "completions/mean_length": 216.875, "completions/mean_terminated_length": 216.875, "completions/min_length": 214.0, "completions/min_terminated_length": 214.0, "entropy": 0.052406568080186844, "epoch": 0.06743784391693779, "frac_reward_zero_std": 0.0, "grad_norm": 1.1392688751220703, "learning_rate": 4.915151515151516e-06, "loss": -0.006, "num_tokens": 3772903.0, "reward": 0.5340501070022583, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9947628974914551, "reward_meter_std": 0.0007815666613169014, "reward_repeat_penalty_mean": 0.6442307829856873, "reward_repeat_penalty_std": 0.039811473339796066, "reward_std": 0.03305771201848984, "reward_total_composite_mean": 0.5340501070022583, "reward_total_composite_std": 0.03305770829319954, "reward_total_mean": 0.5340501070022583, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9947628974914551, "rewards/meter/std": 0.0007815666613169014, "rewards/repeat_penalty/mean": 0.6442307829856873, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.5340501070022583, "rewards/total_composite/std": 0.03305770829319954, "sampling/importance_sampling_ratio/max": 1.3776015043258667, "sampling/importance_sampling_ratio/mean": 1.0005079507827759, "sampling/importance_sampling_ratio/min": 0.32010024785995483, "sampling/sampling_logp_difference/max": 1.1391210556030273, "sampling/sampling_logp_difference/mean": 0.009834333322942257, "step": 1679 }, { "clip_ratio/high_max": 0.0216394632589072, "clip_ratio/high_mean": 0.0216394632589072, "clip_ratio/low_mean": 0.010193796595558524, "clip_ratio/low_min": 0.010193796595558524, "clip_ratio/region_mean": 0.03183325985446572, "completions/clipped_ratio": 0.0, "completions/max_length": 455.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 401.875, "completions/mean_terminated_length": 401.875, "completions/min_length": 373.0, "completions/min_terminated_length": 373.0, "entropy": 0.3997117578983307, "epoch": 0.06747800939872274, "frac_reward_zero_std": 0.0, "grad_norm": 2.291543960571289, "learning_rate": 4.912121212121212e-06, "loss": -0.0168, "num_tokens": 3777950.0, "reward": 0.5898604393005371, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7232142686843872, "reward_count_adherence_std": 0.02525380253791809, "reward_meter_mean": 0.9955718517303467, "reward_meter_std": 0.004510289058089256, "reward_repeat_penalty_mean": 0.8195801973342896, "reward_repeat_penalty_std": 0.1243370845913887, "reward_std": 0.08871299028396606, "reward_total_composite_mean": 0.5898604393005371, "reward_total_composite_std": 0.08871299028396606, "reward_total_mean": 0.5898604393005371, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7232142686843872, "rewards/count_adherence/std": 0.02525380253791809, "rewards/meter/mean": 0.9955718517303467, "rewards/meter/std": 0.004510289058089256, "rewards/repeat_penalty/mean": 0.8195801973342896, "rewards/repeat_penalty/std": 0.1243370845913887, "rewards/total_composite/mean": 0.5898604393005371, "rewards/total_composite/std": 0.08871299028396606, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0117672681808472, "sampling/importance_sampling_ratio/min": 0.17153088748455048, "sampling/sampling_logp_difference/max": 1.780477523803711, "sampling/sampling_logp_difference/mean": 0.0480833500623703, "step": 1680 }, { "clip_ratio/high_max": 0.02016128972172737, "clip_ratio/high_mean": 0.02016128972172737, "clip_ratio/low_mean": 0.008477011695504189, "clip_ratio/low_min": 0.008477011695504189, "clip_ratio/region_mean": 0.02863830141723156, "completions/clipped_ratio": 0.0, "completions/max_length": 31.0, "completions/max_terminated_length": 31.0, "completions/mean_length": 30.25, "completions/mean_terminated_length": 30.25, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.1191922826692462, "epoch": 0.0675181748805077, "frac_reward_zero_std": 0.0, "grad_norm": 4.46686315536499, "learning_rate": 4.90909090909091e-06, "loss": -0.0134, "num_tokens": 3779464.0, "reward": 0.9945183992385864, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9945183992385864, "reward_meter_std": 0.000778909248765558, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007789283408783376, "reward_total_composite_mean": 0.9945183992385864, "reward_total_composite_std": 0.000778909248765558, "reward_total_mean": 0.9945183992385864, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9945183992385864, "rewards/meter/std": 0.000778909248765558, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9945183992385864, "rewards/total_composite/std": 0.000778909248765558, "sampling/importance_sampling_ratio/max": 1.514021873474121, "sampling/importance_sampling_ratio/mean": 0.9981259703636169, "sampling/importance_sampling_ratio/min": 0.5113722085952759, "sampling/sampling_logp_difference/max": 0.6706576347351074, "sampling/sampling_logp_difference/mean": 0.0202788058668375, "step": 1681 }, { "clip_ratio/high_max": 0.0440298137255013, "clip_ratio/high_mean": 0.0440298137255013, "clip_ratio/low_mean": 0.007042253389954567, "clip_ratio/low_min": 0.007042253389954567, "clip_ratio/region_mean": 0.051072067115455866, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 72.5, "completions/mean_terminated_length": 72.5, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.46719325333833694, "epoch": 0.06755834036229265, "frac_reward_zero_std": 0.0, "grad_norm": 7.099659442901611, "learning_rate": 4.906060606060606e-06, "loss": -0.0204, "num_tokens": 3781340.0, "reward": 0.9961867928504944, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961867928504944, "reward_meter_std": 0.0030412215273827314, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003041210351511836, "reward_total_composite_mean": 0.9961867928504944, "reward_total_composite_std": 0.0030412215273827314, "reward_total_mean": 0.9961867928504944, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961867928504944, "rewards/meter/std": 0.0030412215273827314, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961867928504944, "rewards/total_composite/std": 0.0030412215273827314, "sampling/importance_sampling_ratio/max": 1.853277325630188, "sampling/importance_sampling_ratio/mean": 1.0079271793365479, "sampling/importance_sampling_ratio/min": 0.24352942407131195, "sampling/sampling_logp_difference/max": 1.4125175476074219, "sampling/sampling_logp_difference/mean": 0.06393711268901825, "step": 1682 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/region_mean": 0.0019841270986944437, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 63.875, "completions/mean_terminated_length": 63.875, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.01849041902460158, "epoch": 0.0675985058440776, "frac_reward_zero_std": 0.0, "grad_norm": 0.45891278982162476, "learning_rate": 4.903030303030303e-06, "loss": -0.0016, "num_tokens": 3783091.0, "reward": 0.9982389211654663, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982389211654663, "reward_meter_std": 2.360223516006954e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.360223516006954e-05, "reward_total_composite_mean": 0.9982389211654663, "reward_total_composite_std": 2.360223516006954e-05, "reward_total_mean": 0.9982389211654663, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982389211654663, "rewards/meter/std": 2.360223516006954e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982389211654663, "rewards/total_composite/std": 2.360223516006954e-05, "sampling/importance_sampling_ratio/max": 1.0655800104141235, "sampling/importance_sampling_ratio/mean": 1.000044345855713, "sampling/importance_sampling_ratio/min": 0.3951651453971863, "sampling/sampling_logp_difference/max": 0.9284515380859375, "sampling/sampling_logp_difference/mean": 0.004287322983145714, "step": 1683 }, { "clip_ratio/high_max": 0.024855362717062235, "clip_ratio/high_mean": 0.024855362717062235, "clip_ratio/low_mean": 0.007966701406985521, "clip_ratio/low_min": 0.007966701406985521, "clip_ratio/region_mean": 0.032822064124047756, "completions/clipped_ratio": 0.0, "completions/max_length": 180.0, "completions/max_terminated_length": 180.0, "completions/mean_length": 159.5, "completions/mean_terminated_length": 159.5, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.5180013440549374, "epoch": 0.06763867132586256, "frac_reward_zero_std": 0.0, "grad_norm": 3.898653507232666, "learning_rate": 4.9000000000000005e-06, "loss": 0.0107, "num_tokens": 3785743.0, "reward": 0.9769483804702759, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9947781562805176, "reward_meter_std": 0.007610421162098646, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.049505408853292465, "reward_total_composite_mean": 0.9769483804702759, "reward_total_composite_std": 0.049505408853292465, "reward_total_mean": 0.9769483804702759, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9947781562805176, "rewards/meter/std": 0.007610421162098646, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9769483804702759, "rewards/total_composite/std": 0.049505408853292465, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0116267204284668, "sampling/importance_sampling_ratio/min": 0.24394665658473969, "sampling/sampling_logp_difference/max": 1.4108057022094727, "sampling/sampling_logp_difference/mean": 0.059815503656864166, "step": 1684 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.005859375, "clip_ratio/low_min": 0.005859375, "clip_ratio/region_mean": 0.009647253900766373, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 64.25, "completions/mean_terminated_length": 64.25, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.04154683533124626, "epoch": 0.06767883680764751, "frac_reward_zero_std": 0.0, "grad_norm": 2.095104932785034, "learning_rate": 4.896969696969697e-06, "loss": -0.0028, "num_tokens": 3787673.0, "reward": 0.9981728792190552, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981728792190552, "reward_meter_std": 0.0002840534143615514, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00028403810574673116, "reward_total_composite_mean": 0.9981728792190552, "reward_total_composite_std": 0.0002840534143615514, "reward_total_mean": 0.9981728792190552, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981728792190552, "rewards/meter/std": 0.0002840534143615514, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981728792190552, "rewards/total_composite/std": 0.0002840534143615514, "sampling/importance_sampling_ratio/max": 1.5041022300720215, "sampling/importance_sampling_ratio/mean": 0.9983230829238892, "sampling/importance_sampling_ratio/min": 0.1951301395893097, "sampling/sampling_logp_difference/max": 1.6340885162353516, "sampling/sampling_logp_difference/mean": 0.011698653921484947, "step": 1685 }, { "clip_ratio/high_max": 0.02395429043099284, "clip_ratio/high_mean": 0.02395429043099284, "clip_ratio/low_mean": 0.017920455895364285, "clip_ratio/low_min": 0.017920455895364285, "clip_ratio/region_mean": 0.041874746326357126, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 71.75, "completions/mean_terminated_length": 71.75, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.4106294997036457, "epoch": 0.06771900228943246, "frac_reward_zero_std": 0.0, "grad_norm": 4.037059783935547, "learning_rate": 4.893939393939394e-06, "loss": -0.0182, "num_tokens": 3789623.0, "reward": 0.997489333152771, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997489333152771, "reward_meter_std": 0.0010296773398295045, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010296726832166314, "reward_total_composite_mean": 0.997489333152771, "reward_total_composite_std": 0.0010296773398295045, "reward_total_mean": 0.997489333152771, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997489333152771, "rewards/meter/std": 0.0010296773398295045, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997489333152771, "rewards/total_composite/std": 0.0010296773398295045, "sampling/importance_sampling_ratio/max": 1.5631378889083862, "sampling/importance_sampling_ratio/mean": 1.0071872472763062, "sampling/importance_sampling_ratio/min": 0.3446977436542511, "sampling/sampling_logp_difference/max": 1.0650873184204102, "sampling/sampling_logp_difference/mean": 0.04423650726675987, "step": 1686 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.02776301186531782, "epoch": 0.06775916777121742, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.8909090909090914e-06, "loss": 0.0, "num_tokens": 3791023.0, "reward": 0.9977574944496155, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977574944496155, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9977574944496155, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9977574944496155, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977574944496155, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977574944496155, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0424859523773193, "sampling/importance_sampling_ratio/mean": 1.0023516416549683, "sampling/importance_sampling_ratio/min": 0.9605222940444946, "sampling/sampling_logp_difference/max": 0.04160819947719574, "sampling/sampling_logp_difference/mean": 0.0031034064013510942, "step": 1687 }, { "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/low_mean": 0.0070436508394777775, "clip_ratio/low_min": 0.0070436508394777775, "clip_ratio/region_mean": 0.010615079430863261, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 35.5, "completions/mean_terminated_length": 35.5, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.12218342442065477, "epoch": 0.06779933325300237, "frac_reward_zero_std": 0.0, "grad_norm": 4.740636348724365, "learning_rate": 4.887878787878788e-06, "loss": 0.0003, "num_tokens": 3792659.0, "reward": 0.9980167150497437, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980167150497437, "reward_meter_std": 0.00040671354508958757, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004067234694957733, "reward_total_composite_mean": 0.9980167150497437, "reward_total_composite_std": 0.00040671354508958757, "reward_total_mean": 0.9980167150497437, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980167150497437, "rewards/meter/std": 0.00040671354508958757, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980167150497437, "rewards/total_composite/std": 0.00040671354508958757, "sampling/importance_sampling_ratio/max": 1.5504966974258423, "sampling/importance_sampling_ratio/mean": 0.9994746446609497, "sampling/importance_sampling_ratio/min": 0.3956391513347626, "sampling/sampling_logp_difference/max": 0.9272527694702148, "sampling/sampling_logp_difference/mean": 0.021727347746491432, "step": 1688 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.008520788454916328, "epoch": 0.06783949873478733, "frac_reward_zero_std": 0.0, "grad_norm": 0.21065974235534668, "learning_rate": 4.884848484848485e-06, "loss": 0.0024, "num_tokens": 3794387.0, "reward": 0.998286783695221, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998286783695221, "reward_meter_std": 0.00011179452121723443, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011178548447787762, "reward_total_composite_mean": 0.998286783695221, "reward_total_composite_std": 0.00011179452121723443, "reward_total_mean": 0.998286783695221, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998286783695221, "rewards/meter/std": 0.00011179452121723443, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998286783695221, "rewards/total_composite/std": 0.00011179452121723443, "sampling/importance_sampling_ratio/max": 1.028564214706421, "sampling/importance_sampling_ratio/mean": 0.9991194009780884, "sampling/importance_sampling_ratio/min": 0.22980967164039612, "sampling/sampling_logp_difference/max": 1.470503807067871, "sampling/sampling_logp_difference/mean": 0.0037898647133260965, "step": 1689 }, { "clip_ratio/high_max": 0.04497848777100444, "clip_ratio/high_mean": 0.04497848777100444, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/region_mean": 0.048267961479723454, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 36.0, "completions/mean_terminated_length": 36.0, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.3176651671528816, "epoch": 0.06787966421657228, "frac_reward_zero_std": 0.0, "grad_norm": 7.852858066558838, "learning_rate": 4.881818181818182e-06, "loss": 0.0367, "num_tokens": 3795915.0, "reward": 0.934417724609375, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.934417724609375, "reward_meter_std": 0.06754057109355927, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06754057854413986, "reward_total_composite_mean": 0.934417724609375, "reward_total_composite_std": 0.06754057109355927, "reward_total_mean": 0.934417724609375, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.934417724609375, "rewards/meter/std": 0.06754057109355927, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.934417724609375, "rewards/total_composite/std": 0.06754057109355927, "sampling/importance_sampling_ratio/max": 1.7753301858901978, "sampling/importance_sampling_ratio/mean": 1.0148731470108032, "sampling/importance_sampling_ratio/min": 0.3057352602481842, "sampling/sampling_logp_difference/max": 1.1850357055664062, "sampling/sampling_logp_difference/mean": 0.05185196176171303, "step": 1690 }, { "clip_ratio/high_max": 0.025541604205500335, "clip_ratio/high_mean": 0.025541604205500335, "clip_ratio/low_mean": 0.014275046763941646, "clip_ratio/low_min": 0.014275046763941646, "clip_ratio/region_mean": 0.03981665096944198, "completions/clipped_ratio": 0.0, "completions/max_length": 146.0, "completions/max_terminated_length": 146.0, "completions/mean_length": 140.5, "completions/mean_terminated_length": 140.5, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "entropy": 0.28699430637061596, "epoch": 0.06791982969835723, "frac_reward_zero_std": 0.0, "grad_norm": 3.046787738800049, "learning_rate": 4.878787878787879e-06, "loss": -0.0057, "num_tokens": 3798615.0, "reward": 0.824957013130188, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9408473372459412, "reward_meter_std": 0.04903604835271835, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.09155284613370895, "reward_std": 0.11124025285243988, "reward_total_composite_mean": 0.824957013130188, "reward_total_composite_std": 0.11124026030302048, "reward_total_mean": 0.824957013130188, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9408473372459412, "rewards/meter/std": 0.04903604835271835, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.824957013130188, "rewards/total_composite/std": 0.11124026030302048, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.012131690979004, "sampling/importance_sampling_ratio/min": 0.32157430052757263, "sampling/sampling_logp_difference/max": 1.1345267295837402, "sampling/sampling_logp_difference/mean": 0.04019925743341446, "step": 1691 }, { "clip_ratio/high_max": 0.006003972492180765, "clip_ratio/high_mean": 0.006003972492180765, "clip_ratio/low_mean": 0.005935311957728118, "clip_ratio/low_min": 0.005935311957728118, "clip_ratio/region_mean": 0.011939284449908882, "completions/clipped_ratio": 0.0, "completions/max_length": 156.0, "completions/max_terminated_length": 156.0, "completions/mean_length": 148.875, "completions/mean_terminated_length": 148.875, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.03733880212530494, "epoch": 0.06795999518014219, "frac_reward_zero_std": 0.0, "grad_norm": 2.5923614501953125, "learning_rate": 4.875757575757576e-06, "loss": -0.0153, "num_tokens": 3801374.0, "reward": 0.6263105869293213, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9931907653808594, "reward_meter_std": 0.0014422648819163442, "reward_repeat_penalty_mean": 0.6305556297302246, "reward_repeat_penalty_std": 0.0602024681866169, "reward_std": 0.06035333871841431, "reward_total_composite_mean": 0.6263105869293213, "reward_total_composite_std": 0.060353342443704605, "reward_total_mean": 0.6263105869293213, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9931907653808594, "rewards/meter/std": 0.0014422648819163442, "rewards/repeat_penalty/mean": 0.6305556297302246, "rewards/repeat_penalty/std": 0.0602024681866169, "rewards/total_composite/mean": 0.6263105869293213, "rewards/total_composite/std": 0.060353342443704605, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001092791557312, "sampling/importance_sampling_ratio/min": 0.3159392774105072, "sampling/sampling_logp_difference/max": 1.152205228805542, "sampling/sampling_logp_difference/mean": 0.009429208934307098, "step": 1692 }, { "clip_ratio/high_max": 0.02981708408333361, "clip_ratio/high_mean": 0.02981708408333361, "clip_ratio/low_mean": 0.009146409342065454, "clip_ratio/low_min": 0.009146409342065454, "clip_ratio/region_mean": 0.038963493425399065, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 67.25, "completions/mean_terminated_length": 67.25, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.40255092084407806, "epoch": 0.06800016066192714, "frac_reward_zero_std": 0.0, "grad_norm": 5.695197105407715, "learning_rate": 4.872727272727273e-06, "loss": 0.013, "num_tokens": 3803064.0, "reward": 0.7432924509048462, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7432924509048462, "reward_meter_std": 0.2945479154586792, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2945479154586792, "reward_total_composite_mean": 0.7432924509048462, "reward_total_composite_std": 0.2945479154586792, "reward_total_mean": 0.7432924509048462, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7432924509048462, "rewards/meter/std": 0.2945479154586792, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7432924509048462, "rewards/total_composite/std": 0.2945479154586792, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0079994201660156, "sampling/importance_sampling_ratio/min": 0.3200061023235321, "sampling/sampling_logp_difference/max": 1.1394152641296387, "sampling/sampling_logp_difference/mean": 0.05113214999437332, "step": 1693 }, { "clip_ratio/high_max": 0.020258112344890833, "clip_ratio/high_mean": 0.020258112344890833, "clip_ratio/low_mean": 0.010541909374296665, "clip_ratio/low_min": 0.010541909374296665, "clip_ratio/region_mean": 0.030800021719187498, "completions/clipped_ratio": 0.0, "completions/max_length": 182.0, "completions/max_terminated_length": 182.0, "completions/mean_length": 169.375, "completions/mean_terminated_length": 169.375, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.3172535989433527, "epoch": 0.0680403261437121, "frac_reward_zero_std": 0.0, "grad_norm": 2.8474440574645996, "learning_rate": 4.8696969696969705e-06, "loss": 0.0364, "num_tokens": 3805883.0, "reward": 0.6048145294189453, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6835033893585205, "reward_meter_std": 0.2869633436203003, "reward_repeat_penalty_mean": 0.9041666388511658, "reward_repeat_penalty_std": 0.07100624591112137, "reward_std": 0.23514942824840546, "reward_total_composite_mean": 0.6048145294189453, "reward_total_composite_std": 0.23514944314956665, "reward_total_mean": 0.6048145294189453, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6835033893585205, "rewards/meter/std": 0.2869633436203003, "rewards/repeat_penalty/mean": 0.9041666388511658, "rewards/repeat_penalty/std": 0.07100624591112137, "rewards/total_composite/mean": 0.6048145294189453, "rewards/total_composite/std": 0.23514944314956665, "sampling/importance_sampling_ratio/max": 1.8559550046920776, "sampling/importance_sampling_ratio/mean": 1.00909423828125, "sampling/importance_sampling_ratio/min": 0.40250372886657715, "sampling/sampling_logp_difference/max": 0.9100509285926819, "sampling/sampling_logp_difference/mean": 0.03621907904744148, "step": 1694 }, { "clip_ratio/high_max": 0.03078524977900088, "clip_ratio/high_mean": 0.03078524977900088, "clip_ratio/low_mean": 0.006465517450124025, "clip_ratio/low_min": 0.006465517450124025, "clip_ratio/region_mean": 0.037250767229124904, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 60.625, "completions/mean_terminated_length": 60.625, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.17649692203849554, "epoch": 0.06808049162549705, "frac_reward_zero_std": 0.0, "grad_norm": 2.8666982650756836, "learning_rate": 4.866666666666667e-06, "loss": -0.0183, "num_tokens": 3807608.0, "reward": 0.8901118040084839, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8901118040084839, "reward_meter_std": 0.2941354215145111, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2941353917121887, "reward_total_composite_mean": 0.8901118040084839, "reward_total_composite_std": 0.2941354215145111, "reward_total_mean": 0.8901118040084839, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8901118040084839, "rewards/meter/std": 0.2941354215145111, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8901118040084839, "rewards/total_composite/std": 0.2941354215145111, "sampling/importance_sampling_ratio/max": 1.2877650260925293, "sampling/importance_sampling_ratio/mean": 0.9950342178344727, "sampling/importance_sampling_ratio/min": 0.35067400336265564, "sampling/sampling_logp_difference/max": 1.047898292541504, "sampling/sampling_logp_difference/mean": 0.02997700497508049, "step": 1695 }, { "clip_ratio/high_max": 0.03136626980267465, "clip_ratio/high_mean": 0.03136626980267465, "clip_ratio/low_mean": 0.010750595480203629, "clip_ratio/low_min": 0.010750595480203629, "clip_ratio/region_mean": 0.04211686528287828, "completions/clipped_ratio": 0.0, "completions/max_length": 110.0, "completions/max_terminated_length": 110.0, "completions/mean_length": 106.5, "completions/mean_terminated_length": 106.5, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.3774713296443224, "epoch": 0.068120657107282, "frac_reward_zero_std": 0.0, "grad_norm": 3.8953700065612793, "learning_rate": 4.863636363636364e-06, "loss": -0.0017, "num_tokens": 3809772.0, "reward": 0.9039969444274902, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9039969444274902, "reward_meter_std": 0.09514570236206055, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09514573216438293, "reward_total_composite_mean": 0.9039969444274902, "reward_total_composite_std": 0.09514570236206055, "reward_total_mean": 0.9039969444274902, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9039969444274902, "rewards/meter/std": 0.09514570236206055, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9039969444274902, "rewards/total_composite/std": 0.09514570236206055, "sampling/importance_sampling_ratio/max": 1.748896598815918, "sampling/importance_sampling_ratio/mean": 0.9994289875030518, "sampling/importance_sampling_ratio/min": 0.20193234086036682, "sampling/sampling_logp_difference/max": 1.5998225212097168, "sampling/sampling_logp_difference/mean": 0.04417286068201065, "step": 1696 }, { "clip_ratio/high_max": 0.026841548271477222, "clip_ratio/high_mean": 0.026841548271477222, "clip_ratio/low_mean": 0.011194029822945595, "clip_ratio/low_min": 0.011194029822945595, "clip_ratio/region_mean": 0.03803557809442282, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.3504845257848501, "epoch": 0.06816082258906696, "frac_reward_zero_std": 0.0, "grad_norm": 12.74048137664795, "learning_rate": 4.8606060606060615e-06, "loss": 0.0098, "num_tokens": 3811557.0, "reward": 0.7374060153961182, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7374060153961182, "reward_meter_std": 0.2923416793346405, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2923416793346405, "reward_total_composite_mean": 0.7374060153961182, "reward_total_composite_std": 0.2923416793346405, "reward_total_mean": 0.7374060153961182, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7374060153961182, "rewards/meter/std": 0.2923416793346405, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7374060153961182, "rewards/total_composite/std": 0.2923416793346405, "sampling/importance_sampling_ratio/max": 1.9715321063995361, "sampling/importance_sampling_ratio/mean": 1.0157667398452759, "sampling/importance_sampling_ratio/min": 0.43160465359687805, "sampling/sampling_logp_difference/max": 0.840245246887207, "sampling/sampling_logp_difference/mean": 0.051393892616033554, "step": 1697 }, { "clip_ratio/high_max": 0.023344212910160422, "clip_ratio/high_mean": 0.023344212910160422, "clip_ratio/low_mean": 0.01929042232222855, "clip_ratio/low_min": 0.01929042232222855, "clip_ratio/region_mean": 0.04263463523238897, "completions/clipped_ratio": 0.0, "completions/max_length": 476.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 456.625, "completions/mean_terminated_length": 456.625, "completions/min_length": 442.0, "completions/min_terminated_length": 442.0, "entropy": 0.4843989312648773, "epoch": 0.06820098807085191, "frac_reward_zero_std": 0.0, "grad_norm": 2.547250747680664, "learning_rate": 4.857575757575758e-06, "loss": 0.0122, "num_tokens": 3817138.0, "reward": 0.6861064434051514, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7678571343421936, "reward_count_adherence_std": 0.033064987510442734, "reward_meter_mean": 0.9678418636322021, "reward_meter_std": 0.03973681107163429, "reward_repeat_penalty_mean": 0.9239448308944702, "reward_repeat_penalty_std": 0.04794442281126976, "reward_std": 0.04557539150118828, "reward_total_composite_mean": 0.6861064434051514, "reward_total_composite_std": 0.04557538405060768, "reward_total_mean": 0.6861064434051514, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7678571343421936, "rewards/count_adherence/std": 0.033064987510442734, "rewards/meter/mean": 0.9678418636322021, "rewards/meter/std": 0.03973681107163429, "rewards/repeat_penalty/mean": 0.9239448308944702, "rewards/repeat_penalty/std": 0.04794442281126976, "rewards/total_composite/mean": 0.6861064434051514, "rewards/total_composite/std": 0.04557538405060768, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0073450803756714, "sampling/importance_sampling_ratio/min": 0.08908424526453018, "sampling/sampling_logp_difference/max": 2.418172836303711, "sampling/sampling_logp_difference/mean": 0.055158257484436035, "step": 1698 }, { "clip_ratio/high_max": 0.01543669670354575, "clip_ratio/high_mean": 0.01543669670354575, "clip_ratio/low_mean": 0.010468951310031116, "clip_ratio/low_min": 0.010468951310031116, "clip_ratio/region_mean": 0.025905648013576865, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 72.375, "completions/mean_terminated_length": 72.375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.23103010654449463, "epoch": 0.06824115355263687, "frac_reward_zero_std": 0.0, "grad_norm": 3.725172758102417, "learning_rate": 4.854545454545455e-06, "loss": -0.0066, "num_tokens": 3818997.0, "reward": 0.9986300468444824, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986300468444824, "reward_meter_std": 0.0003518488083500415, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00035184345324523747, "reward_total_composite_mean": 0.9986300468444824, "reward_total_composite_std": 0.0003518488083500415, "reward_total_mean": 0.9986300468444824, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986300468444824, "rewards/meter/std": 0.0003518488083500415, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986300468444824, "rewards/total_composite/std": 0.0003518488083500415, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0039602518081665, "sampling/importance_sampling_ratio/min": 0.22722196578979492, "sampling/sampling_logp_difference/max": 1.481827974319458, "sampling/sampling_logp_difference/mean": 0.03500673174858093, "step": 1699 }, { "clip_ratio/high_max": 0.008750928391236812, "clip_ratio/high_mean": 0.008750928391236812, "clip_ratio/low_mean": 0.0007267441833391786, "clip_ratio/low_min": 0.0007267441833391786, "clip_ratio/region_mean": 0.00947767257457599, "completions/clipped_ratio": 0.0, "completions/max_length": 173.0, "completions/max_terminated_length": 173.0, "completions/mean_length": 171.375, "completions/mean_terminated_length": 171.375, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.04710668884217739, "epoch": 0.06828131903442182, "frac_reward_zero_std": 0.0, "grad_norm": 1.9409257173538208, "learning_rate": 4.851515151515152e-06, "loss": 0.0025, "num_tokens": 3822000.0, "reward": 0.6838877201080322, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9947851896286011, "reward_meter_std": 0.0019038140308111906, "reward_repeat_penalty_mean": 0.6875, "reward_repeat_penalty_std": 0.035355325788259506, "reward_std": 0.034654706716537476, "reward_total_composite_mean": 0.6838877201080322, "reward_total_composite_std": 0.034654706716537476, "reward_total_mean": 0.6838877201080322, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9947851896286011, "rewards/meter/std": 0.0019038140308111906, "rewards/repeat_penalty/mean": 0.6875, "rewards/repeat_penalty/std": 0.035355325788259506, "rewards/total_composite/mean": 0.6838877201080322, "rewards/total_composite/std": 0.034654706716537476, "sampling/importance_sampling_ratio/max": 1.4856215715408325, "sampling/importance_sampling_ratio/mean": 0.9983268976211548, "sampling/importance_sampling_ratio/min": 0.0010816717986017466, "sampling/sampling_logp_difference/max": 6.82924747467041, "sampling/sampling_logp_difference/mean": 0.014711027964949608, "step": 1700 }, { "epoch": 0.06828131903442182, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 386.46153846153845, "eval_completions/max_terminated_length": 386.46153846153845, "eval_completions/mean_length": 213.16346153846155, "eval_completions/mean_terminated_length": 213.16346153846155, "eval_completions/min_length": 64.15384615384616, "eval_completions/min_terminated_length": 64.15384615384616, "eval_entropy": 0.20546403641884142, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 3822000.0, "eval_reward": 0.5148563522558945, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.9180220273824838, "eval_reward_count_adherence_std": 0.10367171552318794, "eval_reward_meter_mean": 0.7185193575345553, "eval_reward_meter_std": 0.4064074754714966, "eval_reward_repeat_penalty_mean": 0.7850257295828599, "eval_reward_repeat_penalty_std": 0.15948278972735772, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5148563522558945, "eval_reward_total_composite_std": 0.3435696088350736, "eval_reward_total_mean": 0.5148563522558945, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.9180220273824838, "eval_rewards/count_adherence/std": 0.10367171552318794, "eval_rewards/meter/mean": 0.7185193575345553, "eval_rewards/meter/std": 0.4064074754714966, "eval_rewards/repeat_penalty/mean": 0.7850257295828599, "eval_rewards/repeat_penalty/std": 0.15948278972735772, "eval_rewards/total_composite/mean": 0.5148563522558945, "eval_rewards/total_composite/std": 0.3435696088350736, "eval_runtime": 72.6468, "eval_samples_per_second": 1.432, "eval_sampling/importance_sampling_ratio/max": 1.4555618029374342, "eval_sampling/importance_sampling_ratio/mean": 1.0051932701697717, "eval_sampling/importance_sampling_ratio/min": 0.3100252260382359, "eval_sampling/sampling_logp_difference/max": 1.5070369427020733, "eval_sampling/sampling_logp_difference/mean": 0.020774336352657814, "eval_steps_per_second": 0.179, "step": 1700 }, { "clip_ratio/high_max": 0.02072087489068508, "clip_ratio/high_mean": 0.02072087489068508, "clip_ratio/low_mean": 0.008105989079922438, "clip_ratio/low_min": 0.008105989079922438, "clip_ratio/region_mean": 0.02882686397060752, "completions/clipped_ratio": 0.0, "completions/max_length": 266.0, "completions/max_terminated_length": 266.0, "completions/mean_length": 247.25, "completions/mean_terminated_length": 247.25, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.2732624653726816, "epoch": 0.06832148451620677, "frac_reward_zero_std": 0.0, "grad_norm": 2.8516652584075928, "learning_rate": 4.848484848484849e-06, "loss": 0.0094, "num_tokens": 3825370.0, "reward": 0.6609899997711182, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8367831110954285, "reward_meter_std": 0.21646961569786072, "reward_repeat_penalty_mean": 0.7960164546966553, "reward_repeat_penalty_std": 0.12943987548351288, "reward_std": 0.1882331222295761, "reward_total_composite_mean": 0.6609899997711182, "reward_total_composite_std": 0.1882331371307373, "reward_total_mean": 0.6609899997711182, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8367831110954285, "rewards/meter/std": 0.21646961569786072, "rewards/repeat_penalty/mean": 0.7960164546966553, "rewards/repeat_penalty/std": 0.12943987548351288, "rewards/total_composite/mean": 0.6609899997711182, "rewards/total_composite/std": 0.1882331371307373, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0024902820587158, "sampling/importance_sampling_ratio/min": 6.5306817305099685e-06, "sampling/sampling_logp_difference/max": 11.93899917602539, "sampling/sampling_logp_difference/mean": 0.04899187758564949, "step": 1701 }, { "clip_ratio/high_max": 0.012898168293759227, "clip_ratio/high_mean": 0.012898168293759227, "clip_ratio/low_mean": 0.011560323182493448, "clip_ratio/low_min": 0.011560323182493448, "clip_ratio/region_mean": 0.024458491476252675, "completions/clipped_ratio": 0.0, "completions/max_length": 117.0, "completions/max_terminated_length": 117.0, "completions/mean_length": 108.5, "completions/mean_terminated_length": 108.5, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.24685455113649368, "epoch": 0.06836164999799173, "frac_reward_zero_std": 0.0, "grad_norm": 3.22273325920105, "learning_rate": 4.845454545454546e-06, "loss": 0.013, "num_tokens": 3827574.0, "reward": 0.8985103964805603, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983426332473755, "reward_meter_std": 0.000583991059102118, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.10674867779016495, "reward_total_composite_mean": 0.8985103964805603, "reward_total_composite_std": 0.10674867779016495, "reward_total_mean": 0.8985103964805603, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983426332473755, "rewards/meter/std": 0.000583991059102118, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8985103964805603, "rewards/total_composite/std": 0.10674867779016495, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0079683065414429, "sampling/importance_sampling_ratio/min": 0.2396586388349533, "sampling/sampling_logp_difference/max": 1.428539752960205, "sampling/sampling_logp_difference/mean": 0.03188902884721756, "step": 1702 }, { "clip_ratio/high_max": 0.019598021870478988, "clip_ratio/high_mean": 0.019598021870478988, "clip_ratio/low_mean": 0.017736590933054686, "clip_ratio/low_min": 0.017736590933054686, "clip_ratio/region_mean": 0.03733461280353367, "completions/clipped_ratio": 0.0, "completions/max_length": 352.0, "completions/max_terminated_length": 352.0, "completions/mean_length": 328.125, "completions/mean_terminated_length": 328.125, "completions/min_length": 316.0, "completions/min_terminated_length": 316.0, "entropy": 0.4287557154893875, "epoch": 0.06840181547977668, "frac_reward_zero_std": 0.0, "grad_norm": 1.9336153268814087, "learning_rate": 4.842424242424243e-06, "loss": 0.0086, "num_tokens": 3832143.0, "reward": 0.8237563371658325, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.9899157285690308, "reward_meter_std": 0.008645469322800636, "reward_repeat_penalty_mean": 0.9504464268684387, "reward_repeat_penalty_std": 0.030713941901922226, "reward_std": 0.054851170629262924, "reward_total_composite_mean": 0.8237563371658325, "reward_total_composite_std": 0.054851170629262924, "reward_total_mean": 0.8237563371658325, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.9899157285690308, "rewards/meter/std": 0.008645469322800636, "rewards/repeat_penalty/mean": 0.9504464268684387, "rewards/repeat_penalty/std": 0.030713941901922226, "rewards/total_composite/mean": 0.8237563371658325, "rewards/total_composite/std": 0.054851170629262924, "sampling/importance_sampling_ratio/max": 1.909447431564331, "sampling/importance_sampling_ratio/mean": 1.0071643590927124, "sampling/importance_sampling_ratio/min": 0.12693238258361816, "sampling/sampling_logp_difference/max": 2.064100742340088, "sampling/sampling_logp_difference/mean": 0.05240955203771591, "step": 1703 }, { "clip_ratio/high_max": 0.008767852385062724, "clip_ratio/high_mean": 0.008767852385062724, "clip_ratio/low_mean": 0.002483443822711706, "clip_ratio/low_min": 0.002483443822711706, "clip_ratio/region_mean": 0.01125129620777443, "completions/clipped_ratio": 0.0, "completions/max_length": 151.0, "completions/max_terminated_length": 151.0, "completions/mean_length": 143.125, "completions/mean_terminated_length": 143.125, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.05452722427435219, "epoch": 0.06844198096156164, "frac_reward_zero_std": 0.0, "grad_norm": 4.468354225158691, "learning_rate": 4.83939393939394e-06, "loss": 0.0231, "num_tokens": 3834840.0, "reward": 0.7313905954360962, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9751874804496765, "reward_meter_std": 0.05766330659389496, "reward_repeat_penalty_mean": 0.75, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04324747994542122, "reward_total_composite_mean": 0.7313905954360962, "reward_total_composite_std": 0.04324747994542122, "reward_total_mean": 0.7313905954360962, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9751874804496765, "rewards/meter/std": 0.05766330659389496, "rewards/repeat_penalty/mean": 0.75, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7313905954360962, "rewards/total_composite/std": 0.04324747994542122, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0032126903533936, "sampling/importance_sampling_ratio/min": 0.20562252402305603, "sampling/sampling_logp_difference/max": 1.5817131996154785, "sampling/sampling_logp_difference/mean": 0.013407315127551556, "step": 1704 }, { "clip_ratio/high_max": 0.033178919460624456, "clip_ratio/high_mean": 0.033178919460624456, "clip_ratio/low_mean": 0.02736378228291869, "clip_ratio/low_min": 0.02736378228291869, "clip_ratio/region_mean": 0.06054270174354315, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 75.875, "completions/mean_terminated_length": 75.875, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.66333282366395, "epoch": 0.06848214644334659, "frac_reward_zero_std": 0.0, "grad_norm": 9.636635780334473, "learning_rate": 4.836363636363637e-06, "loss": 0.0492, "num_tokens": 3836711.0, "reward": 0.9914894104003906, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9914894104003906, "reward_meter_std": 0.009747824631631374, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009747819975018501, "reward_total_composite_mean": 0.9914894104003906, "reward_total_composite_std": 0.009747824631631374, "reward_total_mean": 0.9914894104003906, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9914894104003906, "rewards/meter/std": 0.009747824631631374, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9914894104003906, "rewards/total_composite/std": 0.009747824631631374, "sampling/importance_sampling_ratio/max": 1.6591025590896606, "sampling/importance_sampling_ratio/mean": 1.0037219524383545, "sampling/importance_sampling_ratio/min": 0.290093332529068, "sampling/sampling_logp_difference/max": 1.237552523612976, "sampling/sampling_logp_difference/mean": 0.07412047684192657, "step": 1705 }, { "clip_ratio/high_max": 0.0056348692160099745, "clip_ratio/high_mean": 0.0056348692160099745, "clip_ratio/low_mean": 0.007105766620952636, "clip_ratio/low_min": 0.007105766620952636, "clip_ratio/region_mean": 0.01274063583696261, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 302.25, "completions/mean_terminated_length": 302.25, "completions/min_length": 279.0, "completions/min_terminated_length": 279.0, "entropy": 0.13442990742623806, "epoch": 0.06852231192513154, "frac_reward_zero_std": 0.0, "grad_norm": 1.8596045970916748, "learning_rate": 4.833333333333333e-06, "loss": -0.016, "num_tokens": 3840705.0, "reward": 0.6033692955970764, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998089075088501, "reward_meter_std": 0.0013965349644422531, "reward_repeat_penalty_mean": 0.604687511920929, "reward_repeat_penalty_std": 0.15109451115131378, "reward_std": 0.1497408151626587, "reward_total_composite_mean": 0.6033692955970764, "reward_total_composite_std": 0.14974084496498108, "reward_total_mean": 0.6033692955970764, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998089075088501, "rewards/meter/std": 0.0013965349644422531, "rewards/repeat_penalty/mean": 0.604687511920929, "rewards/repeat_penalty/std": 0.15109451115131378, "rewards/total_composite/mean": 0.6033692955970764, "rewards/total_composite/std": 0.14974084496498108, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016717910766602, "sampling/importance_sampling_ratio/min": 0.0005643144831992686, "sampling/sampling_logp_difference/max": 7.479898929595947, "sampling/sampling_logp_difference/mean": 0.022071924060583115, "step": 1706 }, { "clip_ratio/high_max": 0.014213158516213298, "clip_ratio/high_mean": 0.014213158516213298, "clip_ratio/low_mean": 0.004166666883975267, "clip_ratio/low_min": 0.004166666883975267, "clip_ratio/region_mean": 0.018379825400188565, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.125, "completions/mean_terminated_length": 61.125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.1731706503778696, "epoch": 0.0685624774069165, "frac_reward_zero_std": 0.0, "grad_norm": 4.459820747375488, "learning_rate": 4.830303030303031e-06, "loss": 0.0035, "num_tokens": 3842546.0, "reward": 0.9614025354385376, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9614025354385376, "reward_meter_std": 0.046098750084638596, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.046098742634058, "reward_total_composite_mean": 0.9614025354385376, "reward_total_composite_std": 0.046098750084638596, "reward_total_mean": 0.9614025354385376, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9614025354385376, "rewards/meter/std": 0.046098750084638596, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9614025354385376, "rewards/total_composite/std": 0.046098750084638596, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005699872970581, "sampling/importance_sampling_ratio/min": 0.5145033597946167, "sampling/sampling_logp_difference/max": 0.7878568172454834, "sampling/sampling_logp_difference/mean": 0.0242039505392313, "step": 1707 }, { "clip_ratio/high_max": 0.019748708698898554, "clip_ratio/high_mean": 0.019748708698898554, "clip_ratio/low_mean": 0.011602062964811921, "clip_ratio/low_min": 0.011602062964811921, "clip_ratio/region_mean": 0.031350771663710475, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 75.75, "completions/mean_terminated_length": 75.75, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.360767001286149, "epoch": 0.06860264288870145, "frac_reward_zero_std": 0.0, "grad_norm": 4.370044231414795, "learning_rate": 4.827272727272728e-06, "loss": -0.0028, "num_tokens": 3844472.0, "reward": 0.9969611167907715, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969611167907715, "reward_meter_std": 0.0019538968335837126, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001953905913978815, "reward_total_composite_mean": 0.9969611167907715, "reward_total_composite_std": 0.0019538968335837126, "reward_total_mean": 0.9969611167907715, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969611167907715, "rewards/meter/std": 0.0019538968335837126, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969611167907715, "rewards/total_composite/std": 0.0019538968335837126, "sampling/importance_sampling_ratio/max": 1.6920305490493774, "sampling/importance_sampling_ratio/mean": 1.0110034942626953, "sampling/importance_sampling_ratio/min": 0.33120596408843994, "sampling/sampling_logp_difference/max": 1.1050148010253906, "sampling/sampling_logp_difference/mean": 0.048442739993333817, "step": 1708 }, { "clip_ratio/high_max": 0.029255983070470393, "clip_ratio/high_mean": 0.029255983070470393, "clip_ratio/low_mean": 0.029091242235153913, "clip_ratio/low_min": 0.029091242235153913, "clip_ratio/region_mean": 0.058347225305624306, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 79.375, "completions/mean_terminated_length": 79.375, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.663462370634079, "epoch": 0.0686428083704864, "frac_reward_zero_std": 0.0, "grad_norm": 7.5510640144348145, "learning_rate": 4.824242424242424e-06, "loss": 0.0066, "num_tokens": 3846403.0, "reward": 0.9834517240524292, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9834517240524292, "reward_meter_std": 0.016320018097758293, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.016320008784532547, "reward_total_composite_mean": 0.9834517240524292, "reward_total_composite_std": 0.016320018097758293, "reward_total_mean": 0.9834517240524292, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9834517240524292, "rewards/meter/std": 0.016320018097758293, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9834517240524292, "rewards/total_composite/std": 0.016320018097758293, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.019029974937439, "sampling/importance_sampling_ratio/min": 0.2331438809633255, "sampling/sampling_logp_difference/max": 1.456099510192871, "sampling/sampling_logp_difference/mean": 0.06894521415233612, "step": 1709 }, { "clip_ratio/high_max": 0.008437702199444175, "clip_ratio/high_mean": 0.008437702199444175, "clip_ratio/low_mean": 0.01390780950896442, "clip_ratio/low_min": 0.01390780950896442, "clip_ratio/region_mean": 0.022345511708408594, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 105.125, "completions/mean_terminated_length": 105.125, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.17343183141201735, "epoch": 0.06868297385227136, "frac_reward_zero_std": 0.0, "grad_norm": 2.7458198070526123, "learning_rate": 4.8212121212121215e-06, "loss": 0.0194, "num_tokens": 3848628.0, "reward": 0.9941129684448242, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9941129684448242, "reward_meter_std": 0.0036593328695744276, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0036593314725905657, "reward_total_composite_mean": 0.9941129684448242, "reward_total_composite_std": 0.0036593328695744276, "reward_total_mean": 0.9941129684448242, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9941129684448242, "rewards/meter/std": 0.0036593328695744276, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9941129684448242, "rewards/total_composite/std": 0.0036593328695744276, "sampling/importance_sampling_ratio/max": 1.5832405090332031, "sampling/importance_sampling_ratio/mean": 1.0030418634414673, "sampling/importance_sampling_ratio/min": 0.3037126362323761, "sampling/sampling_logp_difference/max": 1.1916732788085938, "sampling/sampling_logp_difference/mean": 0.020809238776564598, "step": 1710 }, { "clip_ratio/high_max": 0.010149948066100478, "clip_ratio/high_mean": 0.010149948066100478, "clip_ratio/low_mean": 0.010147797176614404, "clip_ratio/low_min": 0.010147797176614404, "clip_ratio/region_mean": 0.020297745242714882, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 62.0, "completions/mean_terminated_length": 62.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.21017488650977612, "epoch": 0.06872313933405631, "frac_reward_zero_std": 0.0, "grad_norm": 6.050206661224365, "learning_rate": 4.818181818181819e-06, "loss": 0.0035, "num_tokens": 3850396.0, "reward": 0.992821455001831, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.992821455001831, "reward_meter_std": 0.0031417824793606997, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00314177293330431, "reward_total_composite_mean": 0.992821455001831, "reward_total_composite_std": 0.0031417824793606997, "reward_total_mean": 0.992821455001831, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.992821455001831, "rewards/meter/std": 0.0031417824793606997, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992821455001831, "rewards/total_composite/std": 0.0031417824793606997, "sampling/importance_sampling_ratio/max": 1.8755117654800415, "sampling/importance_sampling_ratio/mean": 1.0075486898422241, "sampling/importance_sampling_ratio/min": 0.4229734241962433, "sampling/sampling_logp_difference/max": 0.8604459762573242, "sampling/sampling_logp_difference/mean": 0.028909733518958092, "step": 1711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.011742424685508013, "clip_ratio/low_min": 0.011742424685508013, "clip_ratio/region_mean": 0.011742424685508013, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 31.125, "completions/mean_terminated_length": 31.125, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.138029710855335, "epoch": 0.06876330481584127, "frac_reward_zero_std": 0.0, "grad_norm": 9.417886734008789, "learning_rate": 4.815151515151515e-06, "loss": 0.0252, "num_tokens": 3851781.0, "reward": 0.99363112449646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99363112449646, "reward_meter_std": 0.0030946105252951384, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0030946091283112764, "reward_total_composite_mean": 0.99363112449646, "reward_total_composite_std": 0.0030946105252951384, "reward_total_mean": 0.99363112449646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99363112449646, "rewards/meter/std": 0.0030946105252951384, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99363112449646, "rewards/total_composite/std": 0.0030946105252951384, "sampling/importance_sampling_ratio/max": 1.6804507970809937, "sampling/importance_sampling_ratio/mean": 1.0098103284835815, "sampling/importance_sampling_ratio/min": 0.5000423192977905, "sampling/sampling_logp_difference/max": 0.6930625438690186, "sampling/sampling_logp_difference/mean": 0.01712997443974018, "step": 1712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 86.0, "completions/mean_terminated_length": 86.0, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.007212614000309259, "epoch": 0.06880347029762622, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.8121212121212125e-06, "loss": 0.0, "num_tokens": 3853917.0, "reward": 0.7951493263244629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939366579055786, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7951493263244629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7951493263244629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939366579055786, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7951493263244629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0226460695266724, "sampling/importance_sampling_ratio/mean": 1.0006271600723267, "sampling/importance_sampling_ratio/min": 0.975571870803833, "sampling/sampling_logp_difference/max": 0.02473144233226776, "sampling/sampling_logp_difference/mean": 0.00075820047641173, "step": 1713 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.007174990780185908, "epoch": 0.06884363577941117, "frac_reward_zero_std": 0.0, "grad_norm": 0.25375768542289734, "learning_rate": 4.80909090909091e-06, "loss": 0.0002, "num_tokens": 3855829.0, "reward": 0.9982520937919617, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982520937919617, "reward_meter_std": 1.3676654816663358e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.3667625353264157e-05, "reward_total_composite_mean": 0.9982520937919617, "reward_total_composite_std": 1.3676654816663358e-05, "reward_total_mean": 0.9982520937919617, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982520937919617, "rewards/meter/std": 1.3676654816663358e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982520937919617, "rewards/total_composite/std": 1.3676654816663358e-05, "sampling/importance_sampling_ratio/max": 1.0173074007034302, "sampling/importance_sampling_ratio/mean": 0.9998658895492554, "sampling/importance_sampling_ratio/min": 0.7841194868087769, "sampling/sampling_logp_difference/max": 0.2431938648223877, "sampling/sampling_logp_difference/mean": 0.0014053680934011936, "step": 1714 }, { "clip_ratio/high_max": 0.0024999999441206455, "clip_ratio/high_mean": 0.0024999999441206455, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0024999999441206455, "completions/clipped_ratio": 0.0, "completions/max_length": 50.0, "completions/max_terminated_length": 50.0, "completions/mean_length": 50.0, "completions/mean_terminated_length": 50.0, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.04075565282255411, "epoch": 0.06888380126119613, "frac_reward_zero_std": 0.0, "grad_norm": 0.6869344115257263, "learning_rate": 4.806060606060606e-06, "loss": 0.0006, "num_tokens": 3857509.0, "reward": 0.9374866485595703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9374866485595703, "reward_meter_std": 3.270595334470272e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.270595334470272e-05, "reward_total_composite_mean": 0.9374866485595703, "reward_total_composite_std": 3.270595334470272e-05, "reward_total_mean": 0.9374866485595703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9374866485595703, "rewards/meter/std": 3.270595334470272e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9374866485595703, "rewards/total_composite/std": 3.270595334470272e-05, "sampling/importance_sampling_ratio/max": 1.3050650358200073, "sampling/importance_sampling_ratio/mean": 1.0022356510162354, "sampling/importance_sampling_ratio/min": 0.5548160076141357, "sampling/sampling_logp_difference/max": 0.5891187191009521, "sampling/sampling_logp_difference/mean": 0.006228248123079538, "step": 1715 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.007712913560681045, "epoch": 0.06892396674298108, "frac_reward_zero_std": 0.0, "grad_norm": 1.8621673583984375, "learning_rate": 4.803030303030303e-06, "loss": -0.0025, "num_tokens": 3859301.0, "reward": 0.9931608438491821, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9931608438491821, "reward_meter_std": 0.0010164134437218308, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001016413327306509, "reward_total_composite_mean": 0.9931608438491821, "reward_total_composite_std": 0.0010164134437218308, "reward_total_mean": 0.9931608438491821, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9931608438491821, "rewards/meter/std": 0.0010164134437218308, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931608438491821, "rewards/total_composite/std": 0.0010164134437218308, "sampling/importance_sampling_ratio/max": 1.0107558965682983, "sampling/importance_sampling_ratio/mean": 0.9991500377655029, "sampling/importance_sampling_ratio/min": 0.2865580916404724, "sampling/sampling_logp_difference/max": 1.2498140335083008, "sampling/sampling_logp_difference/mean": 0.003477482357993722, "step": 1716 }, { "clip_ratio/high_max": 0.033190221060067415, "clip_ratio/high_mean": 0.033190221060067415, "clip_ratio/low_mean": 0.007035988033749163, "clip_ratio/low_min": 0.007035988033749163, "clip_ratio/region_mean": 0.04022620909381658, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 403.75, "completions/mean_terminated_length": 403.75, "completions/min_length": 377.0, "completions/min_terminated_length": 377.0, "entropy": 0.4459982290863991, "epoch": 0.06896413222476604, "frac_reward_zero_std": 0.0, "grad_norm": 2.0890252590179443, "learning_rate": 4.800000000000001e-06, "loss": 0.0087, "num_tokens": 3864531.0, "reward": 0.49961453676223755, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.6583333015441895, "reward_count_adherence_std": 0.0235702246427536, "reward_meter_mean": 0.9169327616691589, "reward_meter_std": 0.19917452335357666, "reward_repeat_penalty_mean": 0.9466373920440674, "reward_repeat_penalty_std": 0.04092821478843689, "reward_std": 0.24736955761909485, "reward_total_composite_mean": 0.49961453676223755, "reward_total_composite_std": 0.24736955761909485, "reward_total_mean": 0.49961453676223755, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.6583333015441895, "rewards/count_adherence/std": 0.0235702246427536, "rewards/meter/mean": 0.9169327616691589, "rewards/meter/std": 0.19917452335357666, "rewards/repeat_penalty/mean": 0.9466373920440674, "rewards/repeat_penalty/std": 0.04092821478843689, "rewards/total_composite/mean": 0.49961453676223755, "rewards/total_composite/std": 0.24736955761909485, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0099897384643555, "sampling/importance_sampling_ratio/min": 0.2162899225950241, "sampling/sampling_logp_difference/max": 1.5311355590820312, "sampling/sampling_logp_difference/mean": 0.049914389848709106, "step": 1717 }, { "clip_ratio/high_max": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.001953125, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.009457326959818602, "epoch": 0.06900429770655099, "frac_reward_zero_std": 0.0, "grad_norm": 0.21243786811828613, "learning_rate": 4.796969696969697e-06, "loss": 0.0003, "num_tokens": 3866227.0, "reward": 0.9982569217681885, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982569217681885, "reward_meter_std": 1.7906948414747603e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.7897747966344468e-05, "reward_total_composite_mean": 0.9982569217681885, "reward_total_composite_std": 1.7906948414747603e-05, "reward_total_mean": 0.9982569217681885, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982569217681885, "rewards/meter/std": 1.7906948414747603e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982569217681885, "rewards/total_composite/std": 1.7906948414747603e-05, "sampling/importance_sampling_ratio/max": 1.103933572769165, "sampling/importance_sampling_ratio/mean": 0.9998619556427002, "sampling/importance_sampling_ratio/min": 0.7363636493682861, "sampling/sampling_logp_difference/max": 0.3060312271118164, "sampling/sampling_logp_difference/mean": 0.00201884051784873, "step": 1718 }, { "clip_ratio/high_max": 0.02710696868598461, "clip_ratio/high_mean": 0.02710696868598461, "clip_ratio/low_mean": 0.021380615420639515, "clip_ratio/low_min": 0.021380615420639515, "clip_ratio/region_mean": 0.048487584106624126, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 282.875, "completions/mean_terminated_length": 282.875, "completions/min_length": 273.0, "completions/min_terminated_length": 273.0, "entropy": 0.5307335741817951, "epoch": 0.06904446318833594, "frac_reward_zero_std": 0.0, "grad_norm": 2.303537368774414, "learning_rate": 4.793939393939394e-06, "loss": -0.0103, "num_tokens": 3870642.0, "reward": 0.9256449937820435, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9829399585723877, "reward_meter_std": 0.022622600197792053, "reward_repeat_penalty_mean": 0.942307710647583, "reward_repeat_penalty_std": 0.054392825812101364, "reward_std": 0.04545941576361656, "reward_total_composite_mean": 0.9256449937820435, "reward_total_composite_std": 0.04545941948890686, "reward_total_mean": 0.9256449937820435, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9829399585723877, "rewards/meter/std": 0.022622600197792053, "rewards/repeat_penalty/mean": 0.942307710647583, "rewards/repeat_penalty/std": 0.054392825812101364, "rewards/total_composite/mean": 0.9256449937820435, "rewards/total_composite/std": 0.04545941948890686, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0142277479171753, "sampling/importance_sampling_ratio/min": 0.015890534967184067, "sampling/sampling_logp_difference/max": 4.142031669616699, "sampling/sampling_logp_difference/mean": 0.06017490476369858, "step": 1719 }, { "clip_ratio/high_max": 0.02708326978608966, "clip_ratio/high_mean": 0.02708326978608966, "clip_ratio/low_mean": 0.01499476860044524, "clip_ratio/low_min": 0.01499476860044524, "clip_ratio/region_mean": 0.0420780383865349, "completions/clipped_ratio": 0.0, "completions/max_length": 418.0, "completions/max_terminated_length": 418.0, "completions/mean_length": 400.625, "completions/mean_terminated_length": 400.625, "completions/min_length": 368.0, "completions/min_terminated_length": 368.0, "entropy": 0.5437066555023193, "epoch": 0.0690846286701209, "frac_reward_zero_std": 0.0, "grad_norm": 2.2101805210113525, "learning_rate": 4.790909090909091e-06, "loss": -0.0169, "num_tokens": 3875535.0, "reward": 0.5811381340026855, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6499999761581421, "reward_count_adherence_std": 0.030860668048262596, "reward_meter_mean": 0.9720591306686401, "reward_meter_std": 0.04081534594297409, "reward_repeat_penalty_mean": 0.9199131727218628, "reward_repeat_penalty_std": 0.08860698342323303, "reward_std": 0.06780374050140381, "reward_total_composite_mean": 0.5811381340026855, "reward_total_composite_std": 0.0678037479519844, "reward_total_mean": 0.5811381340026855, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6499999761581421, "rewards/count_adherence/std": 0.030860668048262596, "rewards/meter/mean": 0.9720591306686401, "rewards/meter/std": 0.04081534594297409, "rewards/repeat_penalty/mean": 0.9199131727218628, "rewards/repeat_penalty/std": 0.08860698342323303, "rewards/total_composite/mean": 0.5811381340026855, "rewards/total_composite/std": 0.0678037479519844, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0108215808868408, "sampling/importance_sampling_ratio/min": 0.2521016299724579, "sampling/sampling_logp_difference/max": 1.3779230117797852, "sampling/sampling_logp_difference/mean": 0.05924157425761223, "step": 1720 }, { "clip_ratio/high_max": 0.02464998373761773, "clip_ratio/high_mean": 0.02464998373761773, "clip_ratio/low_mean": 0.013723055250011384, "clip_ratio/low_min": 0.013723055250011384, "clip_ratio/region_mean": 0.038373038987629116, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 72.0, "completions/mean_terminated_length": 72.0, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.3467879444360733, "epoch": 0.06912479415190585, "frac_reward_zero_std": 0.0, "grad_norm": 5.097805023193359, "learning_rate": 4.787878787878788e-06, "loss": 0.0135, "num_tokens": 3877383.0, "reward": 0.9644453525543213, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9644453525543213, "reward_meter_std": 0.02713916078209877, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.027139168232679367, "reward_total_composite_mean": 0.9644453525543213, "reward_total_composite_std": 0.02713916078209877, "reward_total_mean": 0.9644453525543213, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9644453525543213, "rewards/meter/std": 0.02713916078209877, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9644453525543213, "rewards/total_composite/std": 0.02713916078209877, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0127274990081787, "sampling/importance_sampling_ratio/min": 0.31665629148483276, "sampling/sampling_logp_difference/max": 1.1499383449554443, "sampling/sampling_logp_difference/mean": 0.0413968488574028, "step": 1721 }, { "clip_ratio/high_max": 0.04609629465267062, "clip_ratio/high_mean": 0.04609629465267062, "clip_ratio/low_mean": 0.012550130486488342, "clip_ratio/low_min": 0.012550130486488342, "clip_ratio/region_mean": 0.058646425139158964, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 319.375, "completions/mean_terminated_length": 319.375, "completions/min_length": 307.0, "completions/min_terminated_length": 307.0, "entropy": 0.594971913844347, "epoch": 0.0691649596336908, "frac_reward_zero_std": 0.0, "grad_norm": 2.7361412048339844, "learning_rate": 4.784848484848485e-06, "loss": 0.0331, "num_tokens": 3881642.0, "reward": 0.6435312032699585, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7120033502578735, "reward_meter_std": 0.35525479912757874, "reward_repeat_penalty_mean": 0.9166666865348816, "reward_repeat_penalty_std": 0.1321374922990799, "reward_std": 0.32674068212509155, "reward_total_composite_mean": 0.6435312032699585, "reward_total_composite_std": 0.32674068212509155, "reward_total_mean": 0.6435312032699585, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7120033502578735, "rewards/meter/std": 0.35525479912757874, "rewards/repeat_penalty/mean": 0.9166666865348816, "rewards/repeat_penalty/std": 0.1321374922990799, "rewards/total_composite/mean": 0.6435312032699585, "rewards/total_composite/std": 0.32674068212509155, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.011804461479187, "sampling/importance_sampling_ratio/min": 0.17653696238994598, "sampling/sampling_logp_difference/max": 2.4230339527130127, "sampling/sampling_logp_difference/mean": 0.06873981654644012, "step": 1722 }, { "clip_ratio/high_max": 0.028194751124829054, "clip_ratio/high_mean": 0.028194751124829054, "clip_ratio/low_mean": 0.017084942432120442, "clip_ratio/low_min": 0.017084942432120442, "clip_ratio/region_mean": 0.045279693556949496, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 38.625, "completions/mean_terminated_length": 38.625, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.50562334805727, "epoch": 0.06920512511547576, "frac_reward_zero_std": 0.0, "grad_norm": 6.818909645080566, "learning_rate": 4.7818181818181825e-06, "loss": -0.0342, "num_tokens": 3883271.0, "reward": 0.9971029758453369, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971029758453369, "reward_meter_std": 0.0023624140303581953, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002362414263188839, "reward_total_composite_mean": 0.9971029758453369, "reward_total_composite_std": 0.0023624140303581953, "reward_total_mean": 0.9971029758453369, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971029758453369, "rewards/meter/std": 0.0023624140303581953, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971029758453369, "rewards/total_composite/std": 0.0023624140303581953, "sampling/importance_sampling_ratio/max": 1.597930669784546, "sampling/importance_sampling_ratio/mean": 1.012098789215088, "sampling/importance_sampling_ratio/min": 0.354599267244339, "sampling/sampling_logp_difference/max": 1.0367670059204102, "sampling/sampling_logp_difference/mean": 0.05782073736190796, "step": 1723 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.00390625, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.01320492452941835, "epoch": 0.06924529059726071, "frac_reward_zero_std": 0.0, "grad_norm": 0.12045075744390488, "learning_rate": 4.77878787878788e-06, "loss": 0.0002, "num_tokens": 3885143.0, "reward": 0.9982585310935974, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982585310935974, "reward_meter_std": 4.937959602102637e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.9387075705453753e-05, "reward_total_composite_mean": 0.9982585310935974, "reward_total_composite_std": 4.937959602102637e-05, "reward_total_mean": 0.9982585310935974, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982585310935974, "rewards/meter/std": 4.937959602102637e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982585310935974, "rewards/total_composite/std": 4.937959602102637e-05, "sampling/importance_sampling_ratio/max": 1.6037458181381226, "sampling/importance_sampling_ratio/mean": 1.001138687133789, "sampling/importance_sampling_ratio/min": 0.6142910122871399, "sampling/sampling_logp_difference/max": 0.4872865676879883, "sampling/sampling_logp_difference/mean": 0.0029545121360570192, "step": 1724 }, { "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/region_mean": 0.006221415242180228, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 62.5, "completions/mean_terminated_length": 62.5, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.06342153111472726, "epoch": 0.06928545607904567, "frac_reward_zero_std": 0.0, "grad_norm": 3.2427947521209717, "learning_rate": 4.775757575757576e-06, "loss": -0.0117, "num_tokens": 3886827.0, "reward": 0.9951238632202148, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9951238632202148, "reward_meter_std": 0.0010399873135611415, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010399814927950501, "reward_total_composite_mean": 0.9951238632202148, "reward_total_composite_std": 0.0010399873135611415, "reward_total_mean": 0.9951238632202148, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9951238632202148, "rewards/meter/std": 0.0010399873135611415, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951238632202148, "rewards/total_composite/std": 0.0010399873135611415, "sampling/importance_sampling_ratio/max": 1.5123201608657837, "sampling/importance_sampling_ratio/mean": 1.002244234085083, "sampling/importance_sampling_ratio/min": 0.4437049925327301, "sampling/sampling_logp_difference/max": 0.8125953674316406, "sampling/sampling_logp_difference/mean": 0.009362027049064636, "step": 1725 }, { "clip_ratio/high_max": 0.009680354851298034, "clip_ratio/high_mean": 0.009680354851298034, "clip_ratio/low_mean": 0.011303507490083575, "clip_ratio/low_min": 0.011303507490083575, "clip_ratio/region_mean": 0.02098386234138161, "completions/clipped_ratio": 0.0, "completions/max_length": 136.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 130.125, "completions/mean_terminated_length": 130.125, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.264286732301116, "epoch": 0.06932562156083062, "frac_reward_zero_std": 0.0, "grad_norm": 3.5299017429351807, "learning_rate": 4.772727272727273e-06, "loss": 0.0135, "num_tokens": 3889004.0, "reward": 0.48184216022491455, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5966504812240601, "reward_meter_std": 0.2748834788799286, "reward_repeat_penalty_mean": 0.8214285373687744, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.2112974226474762, "reward_total_composite_mean": 0.48184216022491455, "reward_total_composite_std": 0.2112974375486374, "reward_total_mean": 0.48184216022491455, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5966504812240601, "rewards/meter/std": 0.2748834788799286, "rewards/repeat_penalty/mean": 0.8214285373687744, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.48184216022491455, "rewards/total_composite/std": 0.2112974375486374, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0085442066192627, "sampling/importance_sampling_ratio/min": 0.034789059311151505, "sampling/sampling_logp_difference/max": 3.358452320098877, "sampling/sampling_logp_difference/mean": 0.03716103360056877, "step": 1726 }, { "clip_ratio/high_max": 0.011004327097907662, "clip_ratio/high_mean": 0.011004327097907662, "clip_ratio/low_mean": 0.005514706019312143, "clip_ratio/low_min": 0.005514706019312143, "clip_ratio/region_mean": 0.016519033117219806, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 68.5, "completions/mean_terminated_length": 68.5, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.11760625522583723, "epoch": 0.06936578704261558, "frac_reward_zero_std": 0.0, "grad_norm": 1.0005321502685547, "learning_rate": 4.769696969696971e-06, "loss": -0.0028, "num_tokens": 3890696.0, "reward": 0.9954667687416077, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9954667687416077, "reward_meter_std": 0.004938251804560423, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004938257858157158, "reward_total_composite_mean": 0.9954667687416077, "reward_total_composite_std": 0.004938251804560423, "reward_total_mean": 0.9954667687416077, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9954667687416077, "rewards/meter/std": 0.004938251804560423, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9954667687416077, "rewards/total_composite/std": 0.004938251804560423, "sampling/importance_sampling_ratio/max": 1.4678758382797241, "sampling/importance_sampling_ratio/mean": 1.0034555196762085, "sampling/importance_sampling_ratio/min": 0.5163331627845764, "sampling/sampling_logp_difference/max": 0.6610031127929688, "sampling/sampling_logp_difference/mean": 0.01310752984136343, "step": 1727 }, { "clip_ratio/high_max": 0.041200052946805954, "clip_ratio/high_mean": 0.041200052946805954, "clip_ratio/low_mean": 0.005952381179668009, "clip_ratio/low_min": 0.005952381179668009, "clip_ratio/region_mean": 0.04715243412647396, "completions/clipped_ratio": 0.0, "completions/max_length": 86.0, "completions/max_terminated_length": 86.0, "completions/mean_length": 82.125, "completions/mean_terminated_length": 82.125, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.5294598899781704, "epoch": 0.06940595252440053, "frac_reward_zero_std": 0.0, "grad_norm": 5.639673709869385, "learning_rate": 4.766666666666667e-06, "loss": 0.0319, "num_tokens": 3892529.0, "reward": 0.9778019189834595, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9778019189834595, "reward_meter_std": 0.024416539818048477, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.024416528642177582, "reward_total_composite_mean": 0.9778019189834595, "reward_total_composite_std": 0.024416539818048477, "reward_total_mean": 0.9778019189834595, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9778019189834595, "rewards/meter/std": 0.024416539818048477, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9778019189834595, "rewards/total_composite/std": 0.024416539818048477, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0134650468826294, "sampling/importance_sampling_ratio/min": 0.31155264377593994, "sampling/sampling_logp_difference/max": 1.5115740299224854, "sampling/sampling_logp_difference/mean": 0.055459219962358475, "step": 1728 }, { "clip_ratio/high_max": 0.01018475356977433, "clip_ratio/high_mean": 0.01018475356977433, "clip_ratio/low_mean": 0.0008223684271797538, "clip_ratio/low_min": 0.0008223684271797538, "clip_ratio/region_mean": 0.011007121996954083, "completions/clipped_ratio": 0.0, "completions/max_length": 156.0, "completions/max_terminated_length": 156.0, "completions/mean_length": 150.25, "completions/mean_terminated_length": 150.25, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.09930440969765186, "epoch": 0.06944611800618548, "frac_reward_zero_std": 0.0, "grad_norm": 2.5409841537475586, "learning_rate": 4.763636363636364e-06, "loss": 0.0142, "num_tokens": 3895339.0, "reward": 0.836736798286438, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943567514419556, "reward_meter_std": 0.0023485159035772085, "reward_repeat_penalty_mean": 0.8415178656578064, "reward_repeat_penalty_std": 0.056821081787347794, "reward_std": 0.05605413764715195, "reward_total_composite_mean": 0.836736798286438, "reward_total_composite_std": 0.05605412647128105, "reward_total_mean": 0.836736798286438, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943567514419556, "rewards/meter/std": 0.0023485159035772085, "rewards/repeat_penalty/mean": 0.8415178656578064, "rewards/repeat_penalty/std": 0.056821081787347794, "rewards/total_composite/mean": 0.836736798286438, "rewards/total_composite/std": 0.05605412647128105, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0042328834533691, "sampling/importance_sampling_ratio/min": 0.15576106309890747, "sampling/sampling_logp_difference/max": 1.8594321012496948, "sampling/sampling_logp_difference/mean": 0.016792375594377518, "step": 1729 }, { "clip_ratio/high_max": 0.024344916339032352, "clip_ratio/high_mean": 0.024344916339032352, "clip_ratio/low_mean": 0.013116939691826701, "clip_ratio/low_min": 0.013116939691826701, "clip_ratio/region_mean": 0.03746185603085905, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 131.5, "completions/mean_terminated_length": 131.5, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.14938442967832088, "epoch": 0.06948628348797044, "frac_reward_zero_std": 0.0, "grad_norm": 4.371915340423584, "learning_rate": 4.760606060606061e-06, "loss": 0.0161, "num_tokens": 3897799.0, "reward": 0.6233780384063721, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.9530513882637024, "reward_meter_std": 0.020072180777788162, "reward_repeat_penalty_mean": 0.6852272748947144, "reward_repeat_penalty_std": 0.10432641208171844, "reward_std": 0.12500165402889252, "reward_total_composite_mean": 0.6233780384063721, "reward_total_composite_std": 0.12500165402889252, "reward_total_mean": 0.6233780384063721, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.9530513882637024, "rewards/meter/std": 0.020072180777788162, "rewards/repeat_penalty/mean": 0.6852272748947144, "rewards/repeat_penalty/std": 0.10432641208171844, "rewards/total_composite/mean": 0.6233780384063721, "rewards/total_composite/std": 0.12500165402889252, "sampling/importance_sampling_ratio/max": 1.8074078559875488, "sampling/importance_sampling_ratio/mean": 1.0002104043960571, "sampling/importance_sampling_ratio/min": 0.17804041504859924, "sampling/sampling_logp_difference/max": 1.7257447242736816, "sampling/sampling_logp_difference/mean": 0.030718868598341942, "step": 1730 }, { "clip_ratio/high_max": 0.007273018010891974, "clip_ratio/high_mean": 0.007273018010891974, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.010949488612823188, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.25, "completions/mean_terminated_length": 68.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.08362119551748037, "epoch": 0.06952644896975539, "frac_reward_zero_std": 0.0, "grad_norm": 5.360625267028809, "learning_rate": 4.757575757575758e-06, "loss": -0.0001, "num_tokens": 3899601.0, "reward": 0.996942937374115, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996942937374115, "reward_meter_std": 0.0017319588223472238, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0017319645266979933, "reward_total_composite_mean": 0.996942937374115, "reward_total_composite_std": 0.0017319588223472238, "reward_total_mean": 0.996942937374115, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996942937374115, "rewards/meter/std": 0.0017319588223472238, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996942937374115, "rewards/total_composite/std": 0.0017319588223472238, "sampling/importance_sampling_ratio/max": 1.5452033281326294, "sampling/importance_sampling_ratio/mean": 1.0010026693344116, "sampling/importance_sampling_ratio/min": 0.20893153548240662, "sampling/sampling_logp_difference/max": 1.565748691558838, "sampling/sampling_logp_difference/mean": 0.017641214653849602, "step": 1731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 30.0, "completions/mean_terminated_length": 30.0, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.008589746314100921, "epoch": 0.06956661445154035, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.754545454545455e-06, "loss": 0.0, "num_tokens": 3901161.0, "reward": 0.992271900177002, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.992271900177002, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.992271900177002, "reward_total_composite_std": 0.0, "reward_total_mean": 0.992271900177002, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.992271900177002, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992271900177002, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0209813117980957, "sampling/importance_sampling_ratio/mean": 1.0000195503234863, "sampling/importance_sampling_ratio/min": 0.9235662817955017, "sampling/sampling_logp_difference/max": 0.07951271533966064, "sampling/sampling_logp_difference/mean": 0.0011375559261068702, "step": 1732 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.012850302853621542, "epoch": 0.0696067799333253, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.751515151515152e-06, "loss": 0.0, "num_tokens": 3902961.0, "reward": 0.993520200252533, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.993520200252533, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.993520200252533, "reward_total_composite_std": 0.0, "reward_total_mean": 0.993520200252533, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.993520200252533, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.993520200252533, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0491514205932617, "sampling/importance_sampling_ratio/mean": 1.000504970550537, "sampling/importance_sampling_ratio/min": 0.8947205543518066, "sampling/sampling_logp_difference/max": 0.11124386638402939, "sampling/sampling_logp_difference/mean": 0.0015307251596823335, "step": 1733 }, { "clip_ratio/high_max": 0.01626984216272831, "clip_ratio/high_mean": 0.01626984216272831, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/region_mean": 0.018353175604715943, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 62.25, "completions/mean_terminated_length": 62.25, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.08833573758602142, "epoch": 0.06964694541511025, "frac_reward_zero_std": 0.0, "grad_norm": 2.5078203678131104, "learning_rate": 4.748484848484849e-06, "loss": -0.0114, "num_tokens": 3904691.0, "reward": 0.9855538606643677, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9855538606643677, "reward_meter_std": 0.027445603162050247, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0274455975741148, "reward_total_composite_mean": 0.9855538606643677, "reward_total_composite_std": 0.027445603162050247, "reward_total_mean": 0.9855538606643677, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9855538606643677, "rewards/meter/std": 0.027445603162050247, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9855538606643677, "rewards/total_composite/std": 0.027445603162050247, "sampling/importance_sampling_ratio/max": 1.428898811340332, "sampling/importance_sampling_ratio/mean": 0.9950600266456604, "sampling/importance_sampling_ratio/min": 0.3248794972896576, "sampling/sampling_logp_difference/max": 1.1243009567260742, "sampling/sampling_logp_difference/mean": 0.021761858835816383, "step": 1734 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.008928571827709675, "clip_ratio/low_min": 0.008928571827709675, "clip_ratio/region_mean": 0.008928571827709675, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 56.0, "completions/mean_terminated_length": 56.0, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.037228189525194466, "epoch": 0.06968711089689521, "frac_reward_zero_std": 0.0, "grad_norm": 3.9663925170898438, "learning_rate": 4.745454545454546e-06, "loss": -0.0859, "num_tokens": 3906499.0, "reward": 0.9921239614486694, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9921239614486694, "reward_meter_std": 0.0039491597563028336, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0039491597563028336, "reward_total_composite_mean": 0.9921239614486694, "reward_total_composite_std": 0.0039491597563028336, "reward_total_mean": 0.9921239614486694, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9921239614486694, "rewards/meter/std": 0.0039491597563028336, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9921239614486694, "rewards/total_composite/std": 0.0039491597563028336, "sampling/importance_sampling_ratio/max": 1.7702317237854004, "sampling/importance_sampling_ratio/mean": 1.000370740890503, "sampling/importance_sampling_ratio/min": 0.42796868085861206, "sampling/sampling_logp_difference/max": 0.8487052917480469, "sampling/sampling_logp_difference/mean": 0.006989335175603628, "step": 1735 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/region_mean": 0.0037313431967049837, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 63.125, "completions/mean_terminated_length": 63.125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.12713485257700086, "epoch": 0.06972727637868016, "frac_reward_zero_std": 0.0, "grad_norm": 5.453359127044678, "learning_rate": 4.7424242424242426e-06, "loss": 0.03, "num_tokens": 3908212.0, "reward": 0.8737130761146545, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8737130761146545, "reward_meter_std": 0.3438531160354614, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.34385308623313904, "reward_total_composite_mean": 0.8737130761146545, "reward_total_composite_std": 0.3438531160354614, "reward_total_mean": 0.8737130761146545, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8737130761146545, "rewards/meter/std": 0.3438531160354614, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8737130761146545, "rewards/total_composite/std": 0.3438531160354614, "sampling/importance_sampling_ratio/max": 1.8557262420654297, "sampling/importance_sampling_ratio/mean": 1.003002643585205, "sampling/importance_sampling_ratio/min": 0.3759021461009979, "sampling/sampling_logp_difference/max": 0.978426456451416, "sampling/sampling_logp_difference/mean": 0.01911444216966629, "step": 1736 }, { "clip_ratio/high_max": 0.010484297177754343, "clip_ratio/high_mean": 0.010484297177754343, "clip_ratio/low_mean": 0.008928571944124997, "clip_ratio/low_min": 0.008928571944124997, "clip_ratio/region_mean": 0.01941286912187934, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 129.75, "completions/mean_terminated_length": 129.75, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.18008941877633333, "epoch": 0.06976744186046512, "frac_reward_zero_std": 0.0, "grad_norm": 4.734118461608887, "learning_rate": 4.73939393939394e-06, "loss": -0.0033, "num_tokens": 3910546.0, "reward": 0.6668680310249329, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8132379651069641, "reward_meter_std": 0.2665981948375702, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.09155284613370895, "reward_std": 0.21354439854621887, "reward_total_composite_mean": 0.6668680310249329, "reward_total_composite_std": 0.21354439854621887, "reward_total_mean": 0.6668680310249329, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8132379651069641, "rewards/meter/std": 0.2665981948375702, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.6668680310249329, "rewards/total_composite/std": 0.21354439854621887, "sampling/importance_sampling_ratio/max": 1.6680244207382202, "sampling/importance_sampling_ratio/mean": 1.0038542747497559, "sampling/importance_sampling_ratio/min": 0.3099996745586395, "sampling/sampling_logp_difference/max": 1.1711840629577637, "sampling/sampling_logp_difference/mean": 0.02951814979314804, "step": 1737 }, { "clip_ratio/high_max": 0.0014450866729021072, "clip_ratio/high_mean": 0.0014450866729021072, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0014450866729021072, "completions/clipped_ratio": 0.0, "completions/max_length": 173.0, "completions/max_terminated_length": 173.0, "completions/mean_length": 172.5, "completions/mean_terminated_length": 172.5, "completions/min_length": 169.0, "completions/min_terminated_length": 169.0, "entropy": 0.014313822146505117, "epoch": 0.06980760734225007, "frac_reward_zero_std": 0.0, "grad_norm": 0.2504282295703888, "learning_rate": 4.736363636363637e-06, "loss": 0.0006, "num_tokens": 3913550.0, "reward": 0.7088274955749512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994848370552063, "reward_meter_std": 0.00046264444245025516, "reward_repeat_penalty_mean": 0.7124999761581421, "reward_repeat_penalty_std": 0.0353553481400013, "reward_std": 0.03512410447001457, "reward_total_composite_mean": 0.7088274955749512, "reward_total_composite_std": 0.03512411192059517, "reward_total_mean": 0.7088274955749512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994848370552063, "rewards/meter/std": 0.00046264444245025516, "rewards/repeat_penalty/mean": 0.7124999761581421, "rewards/repeat_penalty/std": 0.0353553481400013, "rewards/total_composite/mean": 0.7088274955749512, "rewards/total_composite/std": 0.03512411192059517, "sampling/importance_sampling_ratio/max": 1.0626802444458008, "sampling/importance_sampling_ratio/mean": 0.9992320537567139, "sampling/importance_sampling_ratio/min": 0.20801043510437012, "sampling/sampling_logp_difference/max": 1.570167064666748, "sampling/sampling_logp_difference/mean": 0.00417359871789813, "step": 1738 }, { "clip_ratio/high_max": 0.028822204330936074, "clip_ratio/high_mean": 0.028822204330936074, "clip_ratio/low_mean": 0.006289557088166475, "clip_ratio/low_min": 0.006289557088166475, "clip_ratio/region_mean": 0.03511176141910255, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 81.75, "completions/mean_terminated_length": 81.75, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.46590081229805946, "epoch": 0.06984777282403502, "frac_reward_zero_std": 0.0, "grad_norm": 4.318221092224121, "learning_rate": 4.7333333333333335e-06, "loss": -0.0102, "num_tokens": 3915644.0, "reward": 0.9966109991073608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9966109991073608, "reward_meter_std": 0.0017567714676260948, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0017567694885656238, "reward_total_composite_mean": 0.9966109991073608, "reward_total_composite_std": 0.0017567714676260948, "reward_total_mean": 0.9966109991073608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9966109991073608, "rewards/meter/std": 0.0017567714676260948, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9966109991073608, "rewards/total_composite/std": 0.0017567714676260948, "sampling/importance_sampling_ratio/max": 1.6328699588775635, "sampling/importance_sampling_ratio/mean": 1.0067728757858276, "sampling/importance_sampling_ratio/min": 0.42967894673347473, "sampling/sampling_logp_difference/max": 0.8447170257568359, "sampling/sampling_logp_difference/mean": 0.04741675779223442, "step": 1739 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.00909090880304575, "clip_ratio/low_min": 0.00909090880304575, "clip_ratio/region_mean": 0.00909090880304575, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 57.5, "completions/mean_terminated_length": 57.5, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.028293918527197093, "epoch": 0.06988793830581998, "frac_reward_zero_std": 0.0, "grad_norm": 7.030989646911621, "learning_rate": 4.730303030303031e-06, "loss": -0.009, "num_tokens": 3917352.0, "reward": 0.8923461437225342, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8923461437225342, "reward_meter_std": 0.2858184278011322, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2858184278011322, "reward_total_composite_mean": 0.8923461437225342, "reward_total_composite_std": 0.2858184278011322, "reward_total_mean": 0.8923461437225342, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8923461437225342, "rewards/meter/std": 0.2858184278011322, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8923461437225342, "rewards/total_composite/std": 0.2858184278011322, "sampling/importance_sampling_ratio/max": 1.3732835054397583, "sampling/importance_sampling_ratio/mean": 1.0057528018951416, "sampling/importance_sampling_ratio/min": 0.8774904012680054, "sampling/sampling_logp_difference/max": 0.3172045946121216, "sampling/sampling_logp_difference/mean": 0.006146615371108055, "step": 1740 }, { "clip_ratio/high_max": 0.004889143165200949, "clip_ratio/high_mean": 0.004889143165200949, "clip_ratio/low_mean": 0.0034997850307263434, "clip_ratio/low_min": 0.0034997850307263434, "clip_ratio/region_mean": 0.008388928195927292, "completions/clipped_ratio": 0.0, "completions/max_length": 290.0, "completions/max_terminated_length": 290.0, "completions/mean_length": 273.875, "completions/mean_terminated_length": 273.875, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "entropy": 0.11419002618640661, "epoch": 0.06992810378760493, "frac_reward_zero_std": 0.0, "grad_norm": 1.3457516431808472, "learning_rate": 4.727272727272728e-06, "loss": 0.0527, "num_tokens": 3920951.0, "reward": 0.5869752168655396, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.910714328289032, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.9979905486106873, "reward_meter_std": 0.0006236601620912552, "reward_repeat_penalty_mean": 0.6429487466812134, "reward_repeat_penalty_std": 0.04667491465806961, "reward_std": 0.08794166892766953, "reward_total_composite_mean": 0.5869752168655396, "reward_total_composite_std": 0.08794166892766953, "reward_total_mean": 0.5869752168655396, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.910714328289032, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.9979905486106873, "rewards/meter/std": 0.0006236601620912552, "rewards/repeat_penalty/mean": 0.6429487466812134, "rewards/repeat_penalty/std": 0.04667491465806961, "rewards/total_composite/mean": 0.5869752168655396, "rewards/total_composite/std": 0.08794166892766953, "sampling/importance_sampling_ratio/max": 1.4954962730407715, "sampling/importance_sampling_ratio/mean": 1.0031613111495972, "sampling/importance_sampling_ratio/min": 0.006186812650412321, "sampling/sampling_logp_difference/max": 5.0853352546691895, "sampling/sampling_logp_difference/mean": 0.01812179759144783, "step": 1741 }, { "clip_ratio/high_max": 0.02945910906419158, "clip_ratio/high_mean": 0.02945910906419158, "clip_ratio/low_mean": 0.0020493179326877, "clip_ratio/low_min": 0.0020493179326877, "clip_ratio/region_mean": 0.03150842699687928, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 119.5, "completions/mean_terminated_length": 119.5, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.397213451564312, "epoch": 0.06996826926938989, "frac_reward_zero_std": 0.0, "grad_norm": 2.8153226375579834, "learning_rate": 4.724242424242424e-06, "loss": 0.0179, "num_tokens": 3923203.0, "reward": 0.9471026659011841, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969746470451355, "reward_meter_std": 0.0006134378490969539, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09206003695726395, "reward_total_composite_mean": 0.9471026659011841, "reward_total_composite_std": 0.09206004440784454, "reward_total_mean": 0.9471026659011841, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969746470451355, "rewards/meter/std": 0.0006134378490969539, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9471026659011841, "rewards/total_composite/std": 0.09206004440784454, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0121161937713623, "sampling/importance_sampling_ratio/min": 0.25306129455566406, "sampling/sampling_logp_difference/max": 1.3741235733032227, "sampling/sampling_logp_difference/mean": 0.04116864502429962, "step": 1742 }, { "clip_ratio/high_max": 0.0010000000474974513, "clip_ratio/high_mean": 0.0010000000474974513, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0010000000474974513, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 115.375, "completions/mean_terminated_length": 115.375, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.009788335533812642, "epoch": 0.07000843475117484, "frac_reward_zero_std": 0.0, "grad_norm": 3.2874183654785156, "learning_rate": 4.721212121212122e-06, "loss": -0.0255, "num_tokens": 3925534.0, "reward": 0.7144180536270142, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939801692962646, "reward_meter_std": 0.0004661441489588469, "reward_repeat_penalty_mean": 0.71875, "reward_repeat_penalty_std": 0.012626901268959045, "reward_std": 0.012203331105411053, "reward_total_composite_mean": 0.7144180536270142, "reward_total_composite_std": 0.012203346937894821, "reward_total_mean": 0.7144180536270142, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939801692962646, "rewards/meter/std": 0.0004661441489588469, "rewards/repeat_penalty/mean": 0.71875, "rewards/repeat_penalty/std": 0.012626901268959045, "rewards/total_composite/mean": 0.7144180536270142, "rewards/total_composite/std": 0.012203346937894821, "sampling/importance_sampling_ratio/max": 1.4329923391342163, "sampling/importance_sampling_ratio/mean": 0.9994580745697021, "sampling/importance_sampling_ratio/min": 0.06089029461145401, "sampling/sampling_logp_difference/max": 2.7986814975738525, "sampling/sampling_logp_difference/mean": 0.005304521415382624, "step": 1743 }, { "clip_ratio/high_max": 0.024856957141309977, "clip_ratio/high_mean": 0.024856957141309977, "clip_ratio/low_mean": 0.011764100869186223, "clip_ratio/low_min": 0.011764100869186223, "clip_ratio/region_mean": 0.0366210580104962, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 116.75, "completions/mean_terminated_length": 116.75, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.44544921815395355, "epoch": 0.07004860023295979, "frac_reward_zero_std": 0.0, "grad_norm": 3.9866888523101807, "learning_rate": 4.718181818181818e-06, "loss": 0.0389, "num_tokens": 3927884.0, "reward": 0.9970704913139343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970704913139343, "reward_meter_std": 0.0010195234790444374, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010194990318268538, "reward_total_composite_mean": 0.9970704913139343, "reward_total_composite_std": 0.0010195234790444374, "reward_total_mean": 0.9970704913139343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970704913139343, "rewards/meter/std": 0.0010195234790444374, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970704913139343, "rewards/total_composite/std": 0.0010195234790444374, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0086452960968018, "sampling/importance_sampling_ratio/min": 0.22523583471775055, "sampling/sampling_logp_difference/max": 1.4906072616577148, "sampling/sampling_logp_difference/mean": 0.05034464970231056, "step": 1744 }, { "clip_ratio/high_max": 0.026507998118177056, "clip_ratio/high_mean": 0.026507998118177056, "clip_ratio/low_mean": 0.01703576883301139, "clip_ratio/low_min": 0.01703576883301139, "clip_ratio/region_mean": 0.043543766951188445, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 71.875, "completions/mean_terminated_length": 71.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.4467601887881756, "epoch": 0.07008876571474475, "frac_reward_zero_std": 0.0, "grad_norm": 5.250048637390137, "learning_rate": 4.715151515151515e-06, "loss": 0.0183, "num_tokens": 3929691.0, "reward": 0.8960655927658081, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8960655927658081, "reward_meter_std": 0.09317556023597717, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09317556023597717, "reward_total_composite_mean": 0.8960655927658081, "reward_total_composite_std": 0.09317556023597717, "reward_total_mean": 0.8960655927658081, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8960655927658081, "rewards/meter/std": 0.09317556023597717, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8960655927658081, "rewards/total_composite/std": 0.09317556023597717, "sampling/importance_sampling_ratio/max": 1.7080049514770508, "sampling/importance_sampling_ratio/mean": 1.0071378946304321, "sampling/importance_sampling_ratio/min": 0.2968732714653015, "sampling/sampling_logp_difference/max": 1.2144498825073242, "sampling/sampling_logp_difference/mean": 0.04669762775301933, "step": 1745 }, { "clip_ratio/high_max": 0.01485425140708685, "clip_ratio/high_mean": 0.01485425140708685, "clip_ratio/low_mean": 0.010288415476679802, "clip_ratio/low_min": 0.010288415476679802, "clip_ratio/region_mean": 0.02514266688376665, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 118.875, "completions/mean_terminated_length": 118.875, "completions/min_length": 114.0, "completions/min_terminated_length": 114.0, "entropy": 0.3711636923253536, "epoch": 0.0701289311965297, "frac_reward_zero_std": 0.0, "grad_norm": 2.7793028354644775, "learning_rate": 4.7121212121212126e-06, "loss": 0.0131, "num_tokens": 3932034.0, "reward": 0.9958000183105469, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958000183105469, "reward_meter_std": 0.0027918845880776644, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0027918978594243526, "reward_total_composite_mean": 0.9958000183105469, "reward_total_composite_std": 0.0027918845880776644, "reward_total_mean": 0.9958000183105469, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958000183105469, "rewards/meter/std": 0.0027918845880776644, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958000183105469, "rewards/total_composite/std": 0.0027918845880776644, "sampling/importance_sampling_ratio/max": 1.7213435173034668, "sampling/importance_sampling_ratio/mean": 1.0113009214401245, "sampling/importance_sampling_ratio/min": 0.3245936334133148, "sampling/sampling_logp_difference/max": 1.1251811981201172, "sampling/sampling_logp_difference/mean": 0.03678244352340698, "step": 1746 }, { "clip_ratio/high_max": 0.00946290313731879, "clip_ratio/high_mean": 0.00946290313731879, "clip_ratio/low_mean": 0.005481442145537585, "clip_ratio/low_min": 0.005481442145537585, "clip_ratio/region_mean": 0.014944345282856375, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 133.5, "completions/mean_terminated_length": 133.5, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.14490803238004446, "epoch": 0.07016909667831465, "frac_reward_zero_std": 0.0, "grad_norm": 2.528543710708618, "learning_rate": 4.709090909090909e-06, "loss": 0.0052, "num_tokens": 3934342.0, "reward": 0.7207491397857666, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9185582995414734, "reward_meter_std": 0.03270983695983887, "reward_repeat_penalty_mean": 0.7857142686843872, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.0627276822924614, "reward_total_composite_mean": 0.7207491397857666, "reward_total_composite_std": 0.0627276748418808, "reward_total_mean": 0.7207491397857666, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9185582995414734, "rewards/meter/std": 0.03270983695983887, "rewards/repeat_penalty/mean": 0.7857142686843872, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.7207491397857666, "rewards/total_composite/std": 0.0627276748418808, "sampling/importance_sampling_ratio/max": 1.7609641551971436, "sampling/importance_sampling_ratio/mean": 1.0059216022491455, "sampling/importance_sampling_ratio/min": 0.36799612641334534, "sampling/sampling_logp_difference/max": 0.9996829032897949, "sampling/sampling_logp_difference/mean": 0.018684184178709984, "step": 1747 }, { "clip_ratio/high_max": 0.02681165118701756, "clip_ratio/high_mean": 0.02681165118701756, "clip_ratio/low_mean": 0.01255695940926671, "clip_ratio/low_min": 0.01255695940926671, "clip_ratio/region_mean": 0.03936861059628427, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 70.625, "completions/mean_terminated_length": 70.625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.45604707673192024, "epoch": 0.07020926216009961, "frac_reward_zero_std": 0.0, "grad_norm": 5.013108253479004, "learning_rate": 4.706060606060606e-06, "loss": 0.0045, "num_tokens": 3936075.0, "reward": 0.9558204412460327, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9558204412460327, "reward_meter_std": 0.03730607032775879, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03730607405304909, "reward_total_composite_mean": 0.9558204412460327, "reward_total_composite_std": 0.03730607032775879, "reward_total_mean": 0.9558204412460327, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9558204412460327, "rewards/meter/std": 0.03730607032775879, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9558204412460327, "rewards/total_composite/std": 0.03730607032775879, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006996989250183, "sampling/importance_sampling_ratio/min": 0.3505965769290924, "sampling/sampling_logp_difference/max": 1.048119068145752, "sampling/sampling_logp_difference/mean": 0.055451519787311554, "step": 1748 }, { "clip_ratio/high_max": 0.048262338852509856, "clip_ratio/high_mean": 0.048262338852509856, "clip_ratio/low_mean": 0.023322351276874542, "clip_ratio/low_min": 0.023322351276874542, "clip_ratio/region_mean": 0.0715846901293844, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 76.375, "completions/mean_terminated_length": 76.375, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.6718878448009491, "epoch": 0.07024942764188456, "frac_reward_zero_std": 0.0, "grad_norm": 6.071586608886719, "learning_rate": 4.7030303030303035e-06, "loss": 0.0118, "num_tokens": 3937998.0, "reward": 0.9766583442687988, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9766583442687988, "reward_meter_std": 0.02674211747944355, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02674211747944355, "reward_total_composite_mean": 0.9766583442687988, "reward_total_composite_std": 0.02674211747944355, "reward_total_mean": 0.9766583442687988, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9766583442687988, "rewards/meter/std": 0.02674211747944355, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9766583442687988, "rewards/total_composite/std": 0.02674211747944355, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0096642971038818, "sampling/importance_sampling_ratio/min": 0.13479571044445038, "sampling/sampling_logp_difference/max": 2.003994941711426, "sampling/sampling_logp_difference/mean": 0.08423186093568802, "step": 1749 }, { "clip_ratio/high_max": 0.021297869854606688, "clip_ratio/high_mean": 0.021297869854606688, "clip_ratio/low_mean": 0.018771880073472857, "clip_ratio/low_min": 0.018771880073472857, "clip_ratio/region_mean": 0.040069749928079545, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 74.75, "completions/mean_terminated_length": 74.75, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.3266411516815424, "epoch": 0.07028959312366952, "frac_reward_zero_std": 0.0, "grad_norm": 2.597658157348633, "learning_rate": 4.7e-06, "loss": -0.0126, "num_tokens": 3939868.0, "reward": 0.998544454574585, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998544454574585, "reward_meter_std": 0.00028903057682327926, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00028902888880111277, "reward_total_composite_mean": 0.998544454574585, "reward_total_composite_std": 0.00028903057682327926, "reward_total_mean": 0.998544454574585, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998544454574585, "rewards/meter/std": 0.00028903057682327926, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998544454574585, "rewards/total_composite/std": 0.00028903057682327926, "sampling/importance_sampling_ratio/max": 1.9856641292572021, "sampling/importance_sampling_ratio/mean": 1.0116918087005615, "sampling/importance_sampling_ratio/min": 0.2113315463066101, "sampling/sampling_logp_difference/max": 1.5543270111083984, "sampling/sampling_logp_difference/mean": 0.04236089065670967, "step": 1750 }, { "epoch": 0.07028959312366952, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 347.38461538461536, "eval_completions/max_terminated_length": 347.38461538461536, "eval_completions/mean_length": 198.26923076923077, "eval_completions/mean_terminated_length": 198.26923076923077, "eval_completions/min_length": 63.15384615384615, "eval_completions/min_terminated_length": 63.15384615384615, "eval_entropy": 0.1877427167044236, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 3939868.0, "eval_reward": 0.5515764791231889, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9028738003510696, "eval_reward_count_adherence_std": 0.12786896412189191, "eval_reward_meter_mean": 0.7541542970217191, "eval_reward_meter_std": 0.3719308634216969, "eval_reward_repeat_penalty_mean": 0.8075039799396808, "eval_reward_repeat_penalty_std": 0.16454027822384468, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5515764791231889, "eval_reward_total_composite_std": 0.33034826585879695, "eval_reward_total_mean": 0.5515764791231889, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9028738003510696, "eval_rewards/count_adherence/std": 0.12786896412189191, "eval_rewards/meter/mean": 0.7541542970217191, "eval_rewards/meter/std": 0.3719308634216969, "eval_rewards/repeat_penalty/mean": 0.8075039799396808, "eval_rewards/repeat_penalty/std": 0.16454027822384468, "eval_rewards/total_composite/mean": 0.5515764791231889, "eval_rewards/total_composite/std": 0.33034826585879695, "eval_runtime": 66.4368, "eval_samples_per_second": 1.565, "eval_sampling/importance_sampling_ratio/max": 1.4309009680381188, "eval_sampling/importance_sampling_ratio/mean": 1.004565248122582, "eval_sampling/importance_sampling_ratio/min": 0.33156666732751405, "eval_sampling/sampling_logp_difference/max": 1.1275318952707143, "eval_sampling/sampling_logp_difference/mean": 0.018031520339158866, "eval_steps_per_second": 0.196, "step": 1750 }, { "clip_ratio/high_max": 0.006016385043039918, "clip_ratio/high_mean": 0.006016385043039918, "clip_ratio/low_mean": 0.00628531095571816, "clip_ratio/low_min": 0.00628531095571816, "clip_ratio/region_mean": 0.012301695998758078, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 59.875, "completions/mean_terminated_length": 59.875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.12970400508493185, "epoch": 0.07032975860545447, "frac_reward_zero_std": 0.0, "grad_norm": 3.8397207260131836, "learning_rate": 4.696969696969698e-06, "loss": -0.0145, "num_tokens": 3941579.0, "reward": 0.994328498840332, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994328498840332, "reward_meter_std": 0.0016656159423291683, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001665610121563077, "reward_total_composite_mean": 0.994328498840332, "reward_total_composite_std": 0.0016656159423291683, "reward_total_mean": 0.994328498840332, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994328498840332, "rewards/meter/std": 0.0016656159423291683, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994328498840332, "rewards/total_composite/std": 0.0016656159423291683, "sampling/importance_sampling_ratio/max": 1.5397275686264038, "sampling/importance_sampling_ratio/mean": 1.0038416385650635, "sampling/importance_sampling_ratio/min": 0.39065343141555786, "sampling/sampling_logp_difference/max": 0.939934492111206, "sampling/sampling_logp_difference/mean": 0.02080354280769825, "step": 1751 }, { "clip_ratio/high_max": 0.04827680857852101, "clip_ratio/high_mean": 0.04827680857852101, "clip_ratio/low_mean": 0.02355617005378008, "clip_ratio/low_min": 0.02355617005378008, "clip_ratio/region_mean": 0.07183297863230109, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 78.25, "completions/mean_terminated_length": 78.25, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.7139590308070183, "epoch": 0.07036992408723942, "frac_reward_zero_std": 0.0, "grad_norm": 9.45784854888916, "learning_rate": 4.693939393939394e-06, "loss": 0.0264, "num_tokens": 3943517.0, "reward": 0.9296289086341858, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9296289086341858, "reward_meter_std": 0.09625125676393509, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09625126421451569, "reward_total_composite_mean": 0.9296289086341858, "reward_total_composite_std": 0.09625125676393509, "reward_total_mean": 0.9296289086341858, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9296289086341858, "rewards/meter/std": 0.09625125676393509, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9296289086341858, "rewards/total_composite/std": 0.09625125676393509, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0267534255981445, "sampling/importance_sampling_ratio/min": 0.230471670627594, "sampling/sampling_logp_difference/max": 1.4676272869110107, "sampling/sampling_logp_difference/mean": 0.08540250360965729, "step": 1752 }, { "clip_ratio/high_max": 0.005952381296083331, "clip_ratio/high_mean": 0.005952381296083331, "clip_ratio/low_mean": 0.008269546087831259, "clip_ratio/low_min": 0.008269546087831259, "clip_ratio/region_mean": 0.01422192738391459, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.375, "completions/mean_terminated_length": 61.375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.11703337356448174, "epoch": 0.07041008956902438, "frac_reward_zero_std": 0.0, "grad_norm": 3.8942646980285645, "learning_rate": 4.690909090909092e-06, "loss": -0.0159, "num_tokens": 3945216.0, "reward": 0.9958833456039429, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958833456039429, "reward_meter_std": 0.0009836297249421477, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009836361277848482, "reward_total_composite_mean": 0.9958833456039429, "reward_total_composite_std": 0.0009836297249421477, "reward_total_mean": 0.9958833456039429, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958833456039429, "rewards/meter/std": 0.0009836297249421477, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958833456039429, "rewards/total_composite/std": 0.0009836297249421477, "sampling/importance_sampling_ratio/max": 1.43393075466156, "sampling/importance_sampling_ratio/mean": 1.0048010349273682, "sampling/importance_sampling_ratio/min": 0.34416088461875916, "sampling/sampling_logp_difference/max": 1.0666460990905762, "sampling/sampling_logp_difference/mean": 0.01771390438079834, "step": 1753 }, { "clip_ratio/high_max": 0.030779469525441527, "clip_ratio/high_mean": 0.030779469525441527, "clip_ratio/low_mean": 0.015182648552581668, "clip_ratio/low_min": 0.015182648552581668, "clip_ratio/region_mean": 0.045962118078023195, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 73.75, "completions/mean_terminated_length": 73.75, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.4345889072865248, "epoch": 0.07045025505080933, "frac_reward_zero_std": 0.0, "grad_norm": 3.0640108585357666, "learning_rate": 4.687878787878788e-06, "loss": 0.0065, "num_tokens": 3947246.0, "reward": 0.998239278793335, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998239278793335, "reward_meter_std": 0.0011535151861608028, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0011535151861608028, "reward_total_composite_mean": 0.998239278793335, "reward_total_composite_std": 0.0011535151861608028, "reward_total_mean": 0.998239278793335, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998239278793335, "rewards/meter/std": 0.0011535151861608028, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998239278793335, "rewards/total_composite/std": 0.0011535151861608028, "sampling/importance_sampling_ratio/max": 1.8171170949935913, "sampling/importance_sampling_ratio/mean": 1.0054799318313599, "sampling/importance_sampling_ratio/min": 0.2711944580078125, "sampling/sampling_logp_difference/max": 1.3049192428588867, "sampling/sampling_logp_difference/mean": 0.050084829330444336, "step": 1754 }, { "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/region_mean": 0.009033613605424762, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.06952884793281555, "epoch": 0.07049042053259429, "frac_reward_zero_std": 0.0, "grad_norm": 1.097916841506958, "learning_rate": 4.684848484848485e-06, "loss": 0.0099, "num_tokens": 3949239.0, "reward": 0.9972657561302185, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972657561302185, "reward_meter_std": 0.0007608237792737782, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007608263404108584, "reward_total_composite_mean": 0.9972657561302185, "reward_total_composite_std": 0.0007608237792737782, "reward_total_mean": 0.9972657561302185, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972657561302185, "rewards/meter/std": 0.0007608237792737782, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972657561302185, "rewards/total_composite/std": 0.0007608237792737782, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0076888799667358, "sampling/importance_sampling_ratio/min": 0.7174111604690552, "sampling/sampling_logp_difference/max": 0.9967951774597168, "sampling/sampling_logp_difference/mean": 0.009788503870368004, "step": 1755 }, { "clip_ratio/high_max": 0.010714285774156451, "clip_ratio/high_mean": 0.010714285774156451, "clip_ratio/low_mean": 0.01736111124046147, "clip_ratio/low_min": 0.01736111124046147, "clip_ratio/region_mean": 0.02807539701461792, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 35.375, "completions/mean_terminated_length": 35.375, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.14098990987986326, "epoch": 0.07053058601437924, "frac_reward_zero_std": 0.0, "grad_norm": 5.3303751945495605, "learning_rate": 4.681818181818183e-06, "loss": 0.013, "num_tokens": 3950770.0, "reward": 0.9960504770278931, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9960504770278931, "reward_meter_std": 0.0025489823892712593, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0025490045081824064, "reward_total_composite_mean": 0.9960504770278931, "reward_total_composite_std": 0.0025489823892712593, "reward_total_mean": 0.9960504770278931, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9960504770278931, "rewards/meter/std": 0.0025489823892712593, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9960504770278931, "rewards/total_composite/std": 0.0025489823892712593, "sampling/importance_sampling_ratio/max": 1.6020153760910034, "sampling/importance_sampling_ratio/mean": 1.0033628940582275, "sampling/importance_sampling_ratio/min": 0.2465495616197586, "sampling/sampling_logp_difference/max": 1.4001922607421875, "sampling/sampling_logp_difference/mean": 0.026823023334145546, "step": 1756 }, { "clip_ratio/high_max": 0.010034454986453056, "clip_ratio/high_mean": 0.010034454986453056, "clip_ratio/low_mean": 0.020188873866572976, "clip_ratio/low_min": 0.020188873866572976, "clip_ratio/region_mean": 0.030223328853026032, "completions/clipped_ratio": 0.0, "completions/max_length": 153.0, "completions/max_terminated_length": 153.0, "completions/mean_length": 148.875, "completions/mean_terminated_length": 148.875, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "entropy": 0.3133071381598711, "epoch": 0.0705707514961642, "frac_reward_zero_std": 0.0, "grad_norm": 2.692962884902954, "learning_rate": 4.678787878787879e-06, "loss": 0.0099, "num_tokens": 3953321.0, "reward": 0.8732737898826599, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979263544082642, "reward_meter_std": 0.0016821667086333036, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.09155284613370895, "reward_std": 0.09220179915428162, "reward_total_composite_mean": 0.8732737898826599, "reward_total_composite_std": 0.09220181405544281, "reward_total_mean": 0.8732737898826599, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979263544082642, "rewards/meter/std": 0.0016821667086333036, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.8732737898826599, "rewards/total_composite/std": 0.09220181405544281, "sampling/importance_sampling_ratio/max": 1.883940577507019, "sampling/importance_sampling_ratio/mean": 1.0055673122406006, "sampling/importance_sampling_ratio/min": 0.3164156377315521, "sampling/sampling_logp_difference/max": 1.1506986618041992, "sampling/sampling_logp_difference/mean": 0.04017876833677292, "step": 1757 }, { "clip_ratio/high_max": 0.03448549238964915, "clip_ratio/high_mean": 0.03448549238964915, "clip_ratio/low_mean": 0.02743737120181322, "clip_ratio/low_min": 0.02743737120181322, "clip_ratio/region_mean": 0.061922863591462374, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.5, "completions/mean_terminated_length": 76.5, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.7143527865409851, "epoch": 0.07061091697794915, "frac_reward_zero_std": 0.0, "grad_norm": 6.611409664154053, "learning_rate": 4.675757575757576e-06, "loss": 0.0195, "num_tokens": 3955117.0, "reward": 0.9615824222564697, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9615824222564697, "reward_meter_std": 0.03308320790529251, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03308321535587311, "reward_total_composite_mean": 0.9615824222564697, "reward_total_composite_std": 0.03308320790529251, "reward_total_mean": 0.9615824222564697, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9615824222564697, "rewards/meter/std": 0.03308320790529251, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9615824222564697, "rewards/total_composite/std": 0.03308320790529251, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0291359424591064, "sampling/importance_sampling_ratio/min": 0.30105844140052795, "sampling/sampling_logp_difference/max": 1.2004508972167969, "sampling/sampling_logp_difference/mean": 0.06566065549850464, "step": 1758 }, { "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/low_mean": 0.016025641234591603, "clip_ratio/low_min": 0.016025641234591603, "clip_ratio/region_mean": 0.019497863482683897, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 36.25, "completions/mean_terminated_length": 36.25, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.12411041604354978, "epoch": 0.0706510824597341, "frac_reward_zero_std": 0.0, "grad_norm": 5.4466166496276855, "learning_rate": 4.6727272727272735e-06, "loss": 0.0438, "num_tokens": 3956703.0, "reward": 0.9933856725692749, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9933856725692749, "reward_meter_std": 0.007373345550149679, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007373336236923933, "reward_total_composite_mean": 0.9933856725692749, "reward_total_composite_std": 0.007373345550149679, "reward_total_mean": 0.9933856725692749, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9933856725692749, "rewards/meter/std": 0.007373345550149679, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9933856725692749, "rewards/total_composite/std": 0.007373345550149679, "sampling/importance_sampling_ratio/max": 1.7787117958068848, "sampling/importance_sampling_ratio/mean": 1.0038979053497314, "sampling/importance_sampling_ratio/min": 0.46098437905311584, "sampling/sampling_logp_difference/max": 0.7743911743164062, "sampling/sampling_logp_difference/mean": 0.021740470081567764, "step": 1759 }, { "clip_ratio/high_max": 0.003289473708719015, "clip_ratio/high_mean": 0.003289473708719015, "clip_ratio/low_mean": 0.0020000000949949026, "clip_ratio/low_min": 0.0020000000949949026, "clip_ratio/region_mean": 0.005289473803713918, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 141.875, "completions/mean_terminated_length": 141.875, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.02644984540529549, "epoch": 0.07069124794151906, "frac_reward_zero_std": 0.0, "grad_norm": 4.831878662109375, "learning_rate": 4.66969696969697e-06, "loss": -0.0871, "num_tokens": 3959470.0, "reward": 0.558111310005188, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_meter_mean": 0.992239236831665, "reward_meter_std": 0.0005570180364884436, "reward_repeat_penalty_mean": 0.609375, "reward_repeat_penalty_std": 0.012938717380166054, "reward_std": 0.051077522337436676, "reward_total_composite_mean": 0.558111310005188, "reward_total_composite_std": 0.05107751861214638, "reward_total_mean": 0.558111310005188, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/meter/mean": 0.992239236831665, "rewards/meter/std": 0.0005570180364884436, "rewards/repeat_penalty/mean": 0.609375, "rewards/repeat_penalty/std": 0.012938717380166054, "rewards/total_composite/mean": 0.558111310005188, "rewards/total_composite/std": 0.05107751861214638, "sampling/importance_sampling_ratio/max": 1.4811135530471802, "sampling/importance_sampling_ratio/mean": 0.9995898008346558, "sampling/importance_sampling_ratio/min": 0.11442553997039795, "sampling/sampling_logp_difference/max": 2.1678309440612793, "sampling/sampling_logp_difference/mean": 0.008833160623908043, "step": 1760 }, { "clip_ratio/high_max": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/region_mean": 0.00390625, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.01341856678482145, "epoch": 0.07073141342330401, "frac_reward_zero_std": 0.0, "grad_norm": 0.03243406489491463, "learning_rate": 4.666666666666667e-06, "loss": -0.0004, "num_tokens": 3961206.0, "reward": 0.9981468319892883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981468319892883, "reward_meter_std": 7.291422207345022e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.28539862393518e-06, "reward_total_composite_mean": 0.9981468319892883, "reward_total_composite_std": 7.291422207345022e-06, "reward_total_mean": 0.9981468319892883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981468319892883, "rewards/meter/std": 7.291422207345022e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981468319892883, "rewards/total_composite/std": 7.291422207345022e-06, "sampling/importance_sampling_ratio/max": 1.4941221475601196, "sampling/importance_sampling_ratio/mean": 1.0011060237884521, "sampling/importance_sampling_ratio/min": 0.8816590905189514, "sampling/sampling_logp_difference/max": 0.4015388488769531, "sampling/sampling_logp_difference/mean": 0.0019470350816845894, "step": 1761 }, { "clip_ratio/high_max": 0.0013157895300537348, "clip_ratio/high_mean": 0.0013157895300537348, "clip_ratio/low_mean": 0.0026178729021921754, "clip_ratio/low_min": 0.0026178729021921754, "clip_ratio/region_mean": 0.00393366243224591, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 95.25, "completions/mean_terminated_length": 95.25, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.04998829588294029, "epoch": 0.07077157890508896, "frac_reward_zero_std": 0.0, "grad_norm": 0.9674805998802185, "learning_rate": 4.663636363636364e-06, "loss": 0.0019, "num_tokens": 3963328.0, "reward": 0.8475098013877869, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970584511756897, "reward_meter_std": 0.00035686453338712454, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09242908656597137, "reward_total_composite_mean": 0.8475098013877869, "reward_total_composite_std": 0.09242907166481018, "reward_total_mean": 0.8475098013877869, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970584511756897, "rewards/meter/std": 0.00035686453338712454, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8475098013877869, "rewards/total_composite/std": 0.09242907166481018, "sampling/importance_sampling_ratio/max": 1.3529959917068481, "sampling/importance_sampling_ratio/mean": 1.0031148195266724, "sampling/importance_sampling_ratio/min": 0.4915185272693634, "sampling/sampling_logp_difference/max": 0.7102556228637695, "sampling/sampling_logp_difference/mean": 0.00739458529278636, "step": 1762 }, { "clip_ratio/high_max": 0.011261471780017018, "clip_ratio/high_mean": 0.011261471780017018, "clip_ratio/low_mean": 0.013588423724286258, "clip_ratio/low_min": 0.013588423724286258, "clip_ratio/region_mean": 0.024849895504303277, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 331.0, "completions/mean_terminated_length": 331.0, "completions/min_length": 316.0, "completions/min_terminated_length": 316.0, "entropy": 0.308476896956563, "epoch": 0.07081174438687392, "frac_reward_zero_std": 0.0, "grad_norm": 1.530092477798462, "learning_rate": 4.660606060606061e-06, "loss": -0.0018, "num_tokens": 3967632.0, "reward": 0.5736789703369141, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.692307710647583, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972841739654541, "reward_meter_std": 0.0010668373433873057, "reward_repeat_penalty_mean": 0.8308823108673096, "reward_repeat_penalty_std": 0.06623479723930359, "reward_std": 0.04592467471957207, "reward_total_composite_mean": 0.5736789703369141, "reward_total_composite_std": 0.04592465981841087, "reward_total_mean": 0.5736789703369141, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.692307710647583, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972841739654541, "rewards/meter/std": 0.0010668373433873057, "rewards/repeat_penalty/mean": 0.8308823108673096, "rewards/repeat_penalty/std": 0.06623479723930359, "rewards/total_composite/mean": 0.5736789703369141, "rewards/total_composite/std": 0.04592465981841087, "sampling/importance_sampling_ratio/max": 1.8286789655685425, "sampling/importance_sampling_ratio/mean": 1.0089269876480103, "sampling/importance_sampling_ratio/min": 0.33526813983917236, "sampling/sampling_logp_difference/max": 1.0928246974945068, "sampling/sampling_logp_difference/mean": 0.030713124200701714, "step": 1763 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.09531538654118776, "epoch": 0.07085190986865887, "frac_reward_zero_std": 0.0, "grad_norm": 5.83869743347168, "learning_rate": 4.657575757575758e-06, "loss": -0.0047, "num_tokens": 3969088.0, "reward": 0.9969139099121094, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969139099121094, "reward_meter_std": 0.002184208482503891, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002184197073802352, "reward_total_composite_mean": 0.9969139099121094, "reward_total_composite_std": 0.002184208482503891, "reward_total_mean": 0.9969139099121094, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969139099121094, "rewards/meter/std": 0.002184208482503891, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969139099121094, "rewards/total_composite/std": 0.002184208482503891, "sampling/importance_sampling_ratio/max": 1.648281455039978, "sampling/importance_sampling_ratio/mean": 0.9999939203262329, "sampling/importance_sampling_ratio/min": 0.40919235348701477, "sampling/sampling_logp_difference/max": 0.8935699462890625, "sampling/sampling_logp_difference/mean": 0.01693633943796158, "step": 1764 }, { "clip_ratio/high_max": 0.00853747595101595, "clip_ratio/high_mean": 0.00853747595101595, "clip_ratio/low_mean": 0.013535281177610159, "clip_ratio/low_min": 0.013535281177610159, "clip_ratio/region_mean": 0.022072757128626108, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 400.625, "completions/mean_terminated_length": 400.625, "completions/min_length": 381.0, "completions/min_terminated_length": 381.0, "entropy": 0.30641947500407696, "epoch": 0.07089207535044383, "frac_reward_zero_std": 0.0, "grad_norm": 1.4083707332611084, "learning_rate": 4.654545454545455e-06, "loss": -0.0026, "num_tokens": 3973789.0, "reward": 0.4966247081756592, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6176470518112183, "reward_count_adherence_std": 0.03144249692559242, "reward_meter_mean": 0.9959809184074402, "reward_meter_std": 0.002331522526219487, "reward_repeat_penalty_mean": 0.8089442253112793, "reward_repeat_penalty_std": 0.07961878180503845, "reward_std": 0.04498036950826645, "reward_total_composite_mean": 0.4966247081756592, "reward_total_composite_std": 0.04498037323355675, "reward_total_mean": 0.4966247081756592, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6176470518112183, "rewards/count_adherence/std": 0.03144249692559242, "rewards/meter/mean": 0.9959809184074402, "rewards/meter/std": 0.002331522526219487, "rewards/repeat_penalty/mean": 0.8089442253112793, "rewards/repeat_penalty/std": 0.07961878180503845, "rewards/total_composite/mean": 0.4966247081756592, "rewards/total_composite/std": 0.04498037323355675, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0090073347091675, "sampling/importance_sampling_ratio/min": 0.2803671956062317, "sampling/sampling_logp_difference/max": 1.2716550827026367, "sampling/sampling_logp_difference/mean": 0.030164016410708427, "step": 1765 }, { "clip_ratio/high_max": 0.03843654436059296, "clip_ratio/high_mean": 0.03843654436059296, "clip_ratio/low_mean": 0.00802111136727035, "clip_ratio/low_min": 0.00802111136727035, "clip_ratio/region_mean": 0.04645765572786331, "completions/clipped_ratio": 0.0, "completions/max_length": 113.0, "completions/max_terminated_length": 113.0, "completions/mean_length": 110.5, "completions/mean_terminated_length": 110.5, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.38782429322600365, "epoch": 0.07093224083222878, "frac_reward_zero_std": 0.0, "grad_norm": 3.6053273677825928, "learning_rate": 4.651515151515152e-06, "loss": 0.0029, "num_tokens": 3976025.0, "reward": 0.9480756521224976, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980085492134094, "reward_meter_std": 0.0012038704007863998, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09206356853246689, "reward_total_composite_mean": 0.9480756521224976, "reward_total_composite_std": 0.09206356108188629, "reward_total_mean": 0.9480756521224976, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980085492134094, "rewards/meter/std": 0.0012038704007863998, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9480756521224976, "rewards/total_composite/std": 0.09206356108188629, "sampling/importance_sampling_ratio/max": 1.5807472467422485, "sampling/importance_sampling_ratio/mean": 1.0053863525390625, "sampling/importance_sampling_ratio/min": 0.14688852429389954, "sampling/sampling_logp_difference/max": 1.918081283569336, "sampling/sampling_logp_difference/mean": 0.04403796046972275, "step": 1766 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.00390625, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.018734007026068866, "epoch": 0.07097240631401373, "frac_reward_zero_std": 0.0, "grad_norm": 0.5792075991630554, "learning_rate": 4.648484848484849e-06, "loss": 0.0007, "num_tokens": 3977817.0, "reward": 0.9981694221496582, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981694221496582, "reward_meter_std": 9.33830306166783e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.336639777757227e-05, "reward_total_composite_mean": 0.9981694221496582, "reward_total_composite_std": 9.33830306166783e-05, "reward_total_mean": 0.9981694221496582, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981694221496582, "rewards/meter/std": 9.33830306166783e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981694221496582, "rewards/total_composite/std": 9.33830306166783e-05, "sampling/importance_sampling_ratio/max": 1.7626099586486816, "sampling/importance_sampling_ratio/mean": 0.9992143511772156, "sampling/importance_sampling_ratio/min": 0.5330914855003357, "sampling/sampling_logp_difference/max": 0.6290621757507324, "sampling/sampling_logp_difference/mean": 0.006093635223805904, "step": 1767 }, { "clip_ratio/high_max": 0.009191176504828036, "clip_ratio/high_mean": 0.009191176504828036, "clip_ratio/low_mean": 0.005514706019312143, "clip_ratio/low_min": 0.005514706019312143, "clip_ratio/region_mean": 0.01470588252414018, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.11503921588882804, "epoch": 0.07101257179579869, "frac_reward_zero_std": 0.0, "grad_norm": 4.614588737487793, "learning_rate": 4.645454545454545e-06, "loss": 0.0077, "num_tokens": 3979649.0, "reward": 0.9972659349441528, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972659349441528, "reward_meter_std": 0.0006626376416534185, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006626351969316602, "reward_total_composite_mean": 0.9972659349441528, "reward_total_composite_std": 0.0006626376416534185, "reward_total_mean": 0.9972659349441528, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972659349441528, "rewards/meter/std": 0.0006626376416534185, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972659349441528, "rewards/total_composite/std": 0.0006626376416534185, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017422437667847, "sampling/importance_sampling_ratio/min": 0.37065815925598145, "sampling/sampling_logp_difference/max": 0.9924750328063965, "sampling/sampling_logp_difference/mean": 0.020455962046980858, "step": 1768 }, { "clip_ratio/high_max": 0.005952381296083331, "clip_ratio/high_mean": 0.005952381296083331, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/region_mean": 0.009920635493472219, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.060156718362122774, "epoch": 0.07105273727758364, "frac_reward_zero_std": 0.0, "grad_norm": 1.5824861526489258, "learning_rate": 4.642424242424243e-06, "loss": 0.0015, "num_tokens": 3981601.0, "reward": 0.9967877864837646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9967877864837646, "reward_meter_std": 0.00022891137632541358, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00022891460685059428, "reward_total_composite_mean": 0.9967877864837646, "reward_total_composite_std": 0.00022891137632541358, "reward_total_mean": 0.9967877864837646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9967877864837646, "rewards/meter/std": 0.00022891137632541358, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967877864837646, "rewards/total_composite/std": 0.00022891137632541358, "sampling/importance_sampling_ratio/max": 1.2106847763061523, "sampling/importance_sampling_ratio/mean": 1.003195881843567, "sampling/importance_sampling_ratio/min": 0.6767785549163818, "sampling/sampling_logp_difference/max": 0.3904110789299011, "sampling/sampling_logp_difference/mean": 0.00849367305636406, "step": 1769 }, { "clip_ratio/high_max": 0.029366800910793245, "clip_ratio/high_mean": 0.029366800910793245, "clip_ratio/low_mean": 0.012420599116012454, "clip_ratio/low_min": 0.012420599116012454, "clip_ratio/region_mean": 0.0417874000268057, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 81.375, "completions/mean_terminated_length": 81.375, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.5354950428009033, "epoch": 0.0710929027593686, "frac_reward_zero_std": 0.0, "grad_norm": 7.558566093444824, "learning_rate": 4.63939393939394e-06, "loss": 0.1371, "num_tokens": 3983404.0, "reward": 0.886616051197052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9489094018936157, "reward_meter_std": 0.09742892533540726, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.18366359174251556, "reward_total_composite_mean": 0.886616051197052, "reward_total_composite_std": 0.18366359174251556, "reward_total_mean": 0.886616051197052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9489094018936157, "rewards/meter/std": 0.09742892533540726, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.886616051197052, "rewards/total_composite/std": 0.18366359174251556, "sampling/importance_sampling_ratio/max": 1.8940720558166504, "sampling/importance_sampling_ratio/mean": 1.0119951963424683, "sampling/importance_sampling_ratio/min": 0.22003860771656036, "sampling/sampling_logp_difference/max": 1.5139522552490234, "sampling/sampling_logp_difference/mean": 0.06127572059631348, "step": 1770 }, { "clip_ratio/high_max": 0.0066069429740309715, "clip_ratio/high_mean": 0.0066069429740309715, "clip_ratio/low_mean": 0.003947368590161204, "clip_ratio/low_min": 0.003947368590161204, "clip_ratio/region_mean": 0.010554311564192176, "completions/clipped_ratio": 0.0, "completions/max_length": 95.0, "completions/max_terminated_length": 95.0, "completions/mean_length": 94.875, "completions/mean_terminated_length": 94.875, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.03504605917260051, "epoch": 0.07113306824115355, "frac_reward_zero_std": 0.0, "grad_norm": 2.8212172985076904, "learning_rate": 4.636363636363636e-06, "loss": 0.006, "num_tokens": 3985659.0, "reward": 0.8732877969741821, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980397820472717, "reward_meter_std": 0.0001069462377927266, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10333773493766785, "reward_total_composite_mean": 0.8732877969741821, "reward_total_composite_std": 0.10333773493766785, "reward_total_mean": 0.8732877969741821, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980397820472717, "rewards/meter/std": 0.0001069462377927266, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8732877969741821, "rewards/total_composite/std": 0.10333773493766785, "sampling/importance_sampling_ratio/max": 1.4593793153762817, "sampling/importance_sampling_ratio/mean": 0.9989795684814453, "sampling/importance_sampling_ratio/min": 0.3786618709564209, "sampling/sampling_logp_difference/max": 0.9711115956306458, "sampling/sampling_logp_difference/mean": 0.009597206488251686, "step": 1771 }, { "clip_ratio/high_max": 0.007936508394777775, "clip_ratio/high_mean": 0.007936508394777775, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/region_mean": 0.010019841836765409, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 62.625, "completions/mean_terminated_length": 62.625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.10218251217156649, "epoch": 0.0711732337229385, "frac_reward_zero_std": 0.0, "grad_norm": 9.031041145324707, "learning_rate": 4.633333333333334e-06, "loss": -0.0013, "num_tokens": 3987488.0, "reward": 0.9955183267593384, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955183267593384, "reward_meter_std": 0.0038252200465649366, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0038252256345003843, "reward_total_composite_mean": 0.9955183267593384, "reward_total_composite_std": 0.0038252200465649366, "reward_total_mean": 0.9955183267593384, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955183267593384, "rewards/meter/std": 0.0038252200465649366, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955183267593384, "rewards/total_composite/std": 0.0038252200465649366, "sampling/importance_sampling_ratio/max": 1.5951910018920898, "sampling/importance_sampling_ratio/mean": 0.9984506964683533, "sampling/importance_sampling_ratio/min": 0.2297336310148239, "sampling/sampling_logp_difference/max": 1.470834732055664, "sampling/sampling_logp_difference/mean": 0.02198421023786068, "step": 1772 }, { "clip_ratio/high_max": 0.005890052416361868, "clip_ratio/high_mean": 0.005890052416361868, "clip_ratio/low_mean": 0.006157553056254983, "clip_ratio/low_min": 0.006157553056254983, "clip_ratio/region_mean": 0.012047605472616851, "completions/clipped_ratio": 0.0, "completions/max_length": 223.0, "completions/max_terminated_length": 223.0, "completions/mean_length": 200.25, "completions/mean_terminated_length": 200.25, "completions/min_length": 191.0, "completions/min_terminated_length": 191.0, "entropy": 0.11565544595941901, "epoch": 0.07121339920472346, "frac_reward_zero_std": 0.0, "grad_norm": 2.2303996086120605, "learning_rate": 4.630303030303031e-06, "loss": 0.0489, "num_tokens": 3990562.0, "reward": 0.6658661961555481, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_meter_mean": 0.9357019662857056, "reward_meter_std": 0.17194922268390656, "reward_repeat_penalty_mean": 0.7526223659515381, "reward_repeat_penalty_std": 0.07428380101919174, "reward_std": 0.11478757858276367, "reward_total_composite_mean": 0.6658661961555481, "reward_total_composite_std": 0.11478758603334427, "reward_total_mean": 0.6658661961555481, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/meter/mean": 0.9357019662857056, "rewards/meter/std": 0.17194922268390656, "rewards/repeat_penalty/mean": 0.7526223659515381, "rewards/repeat_penalty/std": 0.07428380101919174, "rewards/total_composite/mean": 0.6658661961555481, "rewards/total_composite/std": 0.11478758603334427, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023951530456543, "sampling/importance_sampling_ratio/min": 0.053867705166339874, "sampling/sampling_logp_difference/max": 2.9212241172790527, "sampling/sampling_logp_difference/mean": 0.01970170997083187, "step": 1773 }, { "clip_ratio/high_max": 0.0019685039296746254, "clip_ratio/high_mean": 0.0019685039296746254, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0019685039296746254, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 127.125, "completions/mean_terminated_length": 127.125, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.05814863136038184, "epoch": 0.07125356468650841, "frac_reward_zero_std": 0.0, "grad_norm": 2.5007715225219727, "learning_rate": 4.627272727272727e-06, "loss": 0.0009, "num_tokens": 3992891.0, "reward": 0.8370258808135986, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997309684753418, "reward_meter_std": 0.00012155534204794094, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05034033954143524, "reward_total_composite_mean": 0.8370258808135986, "reward_total_composite_std": 0.05034034699201584, "reward_total_mean": 0.8370258808135986, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997309684753418, "rewards/meter/std": 0.00012155534204794094, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8370258808135986, "rewards/total_composite/std": 0.05034034699201584, "sampling/importance_sampling_ratio/max": 1.2224539518356323, "sampling/importance_sampling_ratio/mean": 1.0009863376617432, "sampling/importance_sampling_ratio/min": 0.17306923866271973, "sampling/sampling_logp_difference/max": 1.754063606262207, "sampling/sampling_logp_difference/mean": 0.00933446828275919, "step": 1774 }, { "clip_ratio/high_max": 0.040617201710119843, "clip_ratio/high_mean": 0.040617201710119843, "clip_ratio/low_mean": 0.007042253389954567, "clip_ratio/low_min": 0.007042253389954567, "clip_ratio/region_mean": 0.04765945510007441, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 75.75, "completions/mean_terminated_length": 75.75, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.6292642988264561, "epoch": 0.07129373016829336, "frac_reward_zero_std": 0.0, "grad_norm": 6.361879348754883, "learning_rate": 4.6242424242424245e-06, "loss": -0.0231, "num_tokens": 3994961.0, "reward": 0.9477696418762207, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9477696418762207, "reward_meter_std": 0.10307468473911285, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.10307466983795166, "reward_total_composite_mean": 0.9477696418762207, "reward_total_composite_std": 0.10307468473911285, "reward_total_mean": 0.9477696418762207, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9477696418762207, "rewards/meter/std": 0.10307468473911285, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9477696418762207, "rewards/total_composite/std": 0.10307468473911285, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0081671476364136, "sampling/importance_sampling_ratio/min": 0.036603085696697235, "sampling/sampling_logp_difference/max": 3.3076226711273193, "sampling/sampling_logp_difference/mean": 0.07908248156309128, "step": 1775 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 56.125, "completions/mean_terminated_length": 56.125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.028014210518449545, "epoch": 0.07133389565007832, "frac_reward_zero_std": 0.0, "grad_norm": 9.041850090026855, "learning_rate": 4.621212121212122e-06, "loss": -0.0062, "num_tokens": 3996754.0, "reward": 0.8854995965957642, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8854995965957642, "reward_meter_std": 0.043195273727178574, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04319525882601738, "reward_total_composite_mean": 0.8854995965957642, "reward_total_composite_std": 0.043195273727178574, "reward_total_mean": 0.8854995965957642, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8854995965957642, "rewards/meter/std": 0.043195273727178574, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8854995965957642, "rewards/total_composite/std": 0.043195273727178574, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0033729076385498, "sampling/importance_sampling_ratio/min": 0.7719552516937256, "sampling/sampling_logp_difference/max": 1.9500985145568848, "sampling/sampling_logp_difference/mean": 0.007866953499615192, "step": 1776 }, { "clip_ratio/high_max": 0.03510406916029751, "clip_ratio/high_mean": 0.03510406916029751, "clip_ratio/low_mean": 0.012251984560862184, "clip_ratio/low_min": 0.012251984560862184, "clip_ratio/region_mean": 0.0473560537211597, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.25, "completions/mean_terminated_length": 71.25, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.4883102737367153, "epoch": 0.07137406113186327, "frac_reward_zero_std": 0.0, "grad_norm": 4.8217244148254395, "learning_rate": 4.618181818181818e-06, "loss": 0.013, "num_tokens": 3998604.0, "reward": 0.9572109580039978, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9572109580039978, "reward_meter_std": 0.052123501896858215, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05212350934743881, "reward_total_composite_mean": 0.9572109580039978, "reward_total_composite_std": 0.052123501896858215, "reward_total_mean": 0.9572109580039978, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9572109580039978, "rewards/meter/std": 0.052123501896858215, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9572109580039978, "rewards/total_composite/std": 0.052123501896858215, "sampling/importance_sampling_ratio/max": 1.790594458580017, "sampling/importance_sampling_ratio/mean": 1.0080149173736572, "sampling/importance_sampling_ratio/min": 0.1579430252313614, "sampling/sampling_logp_difference/max": 1.8455209732055664, "sampling/sampling_logp_difference/mean": 0.057501811534166336, "step": 1777 }, { "clip_ratio/high_max": 0.04102262854576111, "clip_ratio/high_mean": 0.04102262854576111, "clip_ratio/low_mean": 0.017992550507187843, "clip_ratio/low_min": 0.017992550507187843, "clip_ratio/region_mean": 0.05901517905294895, "completions/clipped_ratio": 0.0, "completions/max_length": 206.0, "completions/max_terminated_length": 206.0, "completions/mean_length": 160.75, "completions/mean_terminated_length": 160.75, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.7425523065030575, "epoch": 0.07141422661364823, "frac_reward_zero_std": 0.0, "grad_norm": 4.067593097686768, "learning_rate": 4.615151515151515e-06, "loss": 0.0316, "num_tokens": 4001210.0, "reward": 0.7290425300598145, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.7919793128967285, "reward_meter_std": 0.23337964713573456, "reward_repeat_penalty_mean": 0.9665178656578064, "reward_repeat_penalty_std": 0.062180306762456894, "reward_std": 0.19308261573314667, "reward_total_composite_mean": 0.7290425300598145, "reward_total_composite_std": 0.19308261573314667, "reward_total_mean": 0.7290425300598145, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.7919793128967285, "rewards/meter/std": 0.23337964713573456, "rewards/repeat_penalty/mean": 0.9665178656578064, "rewards/repeat_penalty/std": 0.062180306762456894, "rewards/total_composite/mean": 0.7290425300598145, "rewards/total_composite/std": 0.19308261573314667, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.014849066734314, "sampling/importance_sampling_ratio/min": 0.00018584205827210099, "sampling/sampling_logp_difference/max": 8.59061336517334, "sampling/sampling_logp_difference/mean": 0.08566864579916, "step": 1778 }, { "clip_ratio/high_max": 0.0074559845379553735, "clip_ratio/high_mean": 0.0074559845379553735, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/region_mean": 0.010173375892918557, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 134.5, "completions/mean_terminated_length": 134.5, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.052033907268196344, "epoch": 0.07145439209543318, "frac_reward_zero_std": 0.0, "grad_norm": 2.5793981552124023, "learning_rate": 4.612121212121212e-06, "loss": 0.0089, "num_tokens": 4003558.0, "reward": 0.854625403881073, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970629215240479, "reward_meter_std": 0.0009571582195349038, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008204328478313982, "reward_total_composite_mean": 0.854625403881073, "reward_total_composite_std": 0.0008204205660149455, "reward_total_mean": 0.854625403881073, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970629215240479, "rewards/meter/std": 0.0009571582195349038, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.854625403881073, "rewards/total_composite/std": 0.0008204205660149455, "sampling/importance_sampling_ratio/max": 1.5591769218444824, "sampling/importance_sampling_ratio/mean": 1.0006462335586548, "sampling/importance_sampling_ratio/min": 0.44309455156326294, "sampling/sampling_logp_difference/max": 0.8139721155166626, "sampling/sampling_logp_difference/mean": 0.010115544311702251, "step": 1779 }, { "clip_ratio/high_max": 0.018438909435644746, "clip_ratio/high_mean": 0.018438909435644746, "clip_ratio/low_mean": 0.01071572583168745, "clip_ratio/low_min": 0.01071572583168745, "clip_ratio/region_mean": 0.029154635267332196, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 304.125, "completions/mean_terminated_length": 304.125, "completions/min_length": 290.0, "completions/min_terminated_length": 290.0, "entropy": 0.4535849690437317, "epoch": 0.07149455757721813, "frac_reward_zero_std": 0.0, "grad_norm": 2.3572487831115723, "learning_rate": 4.60909090909091e-06, "loss": 0.0069, "num_tokens": 4007791.0, "reward": 0.7979733347892761, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.9952999353408813, "reward_meter_std": 0.0025832315441221, "reward_repeat_penalty_mean": 0.9177083373069763, "reward_repeat_penalty_std": 0.08480106294155121, "reward_std": 0.06831313669681549, "reward_total_composite_mean": 0.7979733347892761, "reward_total_composite_std": 0.06831313669681549, "reward_total_mean": 0.7979733347892761, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.9952999353408813, "rewards/meter/std": 0.0025832315441221, "rewards/repeat_penalty/mean": 0.9177083373069763, "rewards/repeat_penalty/std": 0.08480106294155121, "rewards/total_composite/mean": 0.7979733347892761, "rewards/total_composite/std": 0.06831313669681549, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010035753250122, "sampling/importance_sampling_ratio/min": 0.07349180430173874, "sampling/sampling_logp_difference/max": 2.610581398010254, "sampling/sampling_logp_difference/mean": 0.045330848544836044, "step": 1780 }, { "clip_ratio/high_max": 0.0024999999441206455, "clip_ratio/high_mean": 0.0024999999441206455, "clip_ratio/low_mean": 0.0024999999441206455, "clip_ratio/low_min": 0.0024999999441206455, "clip_ratio/region_mean": 0.004999999888241291, "completions/clipped_ratio": 0.0, "completions/max_length": 50.0, "completions/max_terminated_length": 50.0, "completions/mean_length": 50.0, "completions/mean_terminated_length": 50.0, "completions/min_length": 50.0, "completions/min_terminated_length": 50.0, "entropy": 0.08982300292700529, "epoch": 0.07153472305900309, "frac_reward_zero_std": 0.0, "grad_norm": 2.1961991786956787, "learning_rate": 4.606060606060606e-06, "loss": -0.0006, "num_tokens": 4009447.0, "reward": 0.9375852346420288, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9375852346420288, "reward_meter_std": 0.0008122828439809382, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008122779545374215, "reward_total_composite_mean": 0.9375852346420288, "reward_total_composite_std": 0.0008122828439809382, "reward_total_mean": 0.9375852346420288, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9375852346420288, "rewards/meter/std": 0.0008122828439809382, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9375852346420288, "rewards/total_composite/std": 0.0008122828439809382, "sampling/importance_sampling_ratio/max": 1.5095360279083252, "sampling/importance_sampling_ratio/mean": 1.0008137226104736, "sampling/importance_sampling_ratio/min": 0.4422323405742645, "sampling/sampling_logp_difference/max": 0.815919816493988, "sampling/sampling_logp_difference/mean": 0.011180485598742962, "step": 1781 }, { "clip_ratio/high_max": 0.03219789918512106, "clip_ratio/high_mean": 0.03219789918512106, "clip_ratio/low_mean": 0.004746835213154554, "clip_ratio/low_min": 0.004746835213154554, "clip_ratio/region_mean": 0.036944734398275614, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 80.875, "completions/mean_terminated_length": 80.875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.5954937487840652, "epoch": 0.07157488854078804, "frac_reward_zero_std": 0.0, "grad_norm": 2.5923538208007812, "learning_rate": 4.603030303030304e-06, "loss": -0.0032, "num_tokens": 4011350.0, "reward": 0.8722769021987915, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9967532753944397, "reward_meter_std": 0.0014706900110468268, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3524559438228607, "reward_total_composite_mean": 0.8722769021987915, "reward_total_composite_std": 0.3524559736251831, "reward_total_mean": 0.8722769021987915, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9967532753944397, "rewards/meter/std": 0.0014706900110468268, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8722769021987915, "rewards/total_composite/std": 0.3524559736251831, "sampling/importance_sampling_ratio/max": 1.5346218347549438, "sampling/importance_sampling_ratio/mean": 1.0133930444717407, "sampling/importance_sampling_ratio/min": 0.21322380006313324, "sampling/sampling_logp_difference/max": 1.5454130172729492, "sampling/sampling_logp_difference/mean": 0.05614684522151947, "step": 1782 }, { "clip_ratio/high_max": 0.0008223684271797538, "clip_ratio/high_mean": 0.0008223684271797538, "clip_ratio/low_mean": 0.004111842135898769, "clip_ratio/low_min": 0.004111842135898769, "clip_ratio/region_mean": 0.004934210563078523, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 152.0, "completions/mean_terminated_length": 152.0, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "entropy": 0.024257719283923507, "epoch": 0.071615054022573, "frac_reward_zero_std": 0.0, "grad_norm": 0.8668001294136047, "learning_rate": 4.600000000000001e-06, "loss": 0.0012, "num_tokens": 4014214.0, "reward": 0.6076171398162842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9920245409011841, "reward_meter_std": 0.00013670595944859087, "reward_repeat_penalty_mean": 0.612500011920929, "reward_repeat_penalty_std": 0.035355325788259506, "reward_std": 0.03511865437030792, "reward_total_composite_mean": 0.6076171398162842, "reward_total_composite_std": 0.035118672996759415, "reward_total_mean": 0.6076171398162842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9920245409011841, "rewards/meter/std": 0.00013670595944859087, "rewards/repeat_penalty/mean": 0.612500011920929, "rewards/repeat_penalty/std": 0.035355325788259506, "rewards/total_composite/mean": 0.6076171398162842, "rewards/total_composite/std": 0.035118672996759415, "sampling/importance_sampling_ratio/max": 1.160125970840454, "sampling/importance_sampling_ratio/mean": 0.9984169006347656, "sampling/importance_sampling_ratio/min": 0.09243020415306091, "sampling/sampling_logp_difference/max": 2.3813014030456543, "sampling/sampling_logp_difference/mean": 0.007640105206519365, "step": 1783 }, { "clip_ratio/high_max": 0.02884695027023554, "clip_ratio/high_mean": 0.02884695027023554, "clip_ratio/low_mean": 0.01740056835114956, "clip_ratio/low_min": 0.01740056835114956, "clip_ratio/region_mean": 0.0462475186213851, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.5869800895452499, "epoch": 0.07165521950435795, "frac_reward_zero_std": 0.0, "grad_norm": 4.6875691413879395, "learning_rate": 4.596969696969697e-06, "loss": 0.0087, "num_tokens": 4015926.0, "reward": 0.8023771643638611, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8023771643638611, "reward_meter_std": 0.2034316211938858, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.203431636095047, "reward_total_composite_mean": 0.8023771643638611, "reward_total_composite_std": 0.2034316211938858, "reward_total_mean": 0.8023771643638611, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8023771643638611, "rewards/meter/std": 0.2034316211938858, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8023771643638611, "rewards/total_composite/std": 0.2034316211938858, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0129239559173584, "sampling/importance_sampling_ratio/min": 0.2919274866580963, "sampling/sampling_logp_difference/max": 1.2312498092651367, "sampling/sampling_logp_difference/mean": 0.0649370476603508, "step": 1784 }, { "clip_ratio/high_max": 0.005921379197388887, "clip_ratio/high_mean": 0.005921379197388887, "clip_ratio/low_mean": 0.004167824285104871, "clip_ratio/low_min": 0.004167824285104871, "clip_ratio/region_mean": 0.010089203482493758, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.375, "completions/mean_terminated_length": 62.375, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.08539297711104155, "epoch": 0.0716953849861429, "frac_reward_zero_std": 0.0, "grad_norm": 2.411994457244873, "learning_rate": 4.5939393939393945e-06, "loss": -0.0151, "num_tokens": 4017713.0, "reward": 0.9964926242828369, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9964926242828369, "reward_meter_std": 0.0008941100095398724, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008941145497374237, "reward_total_composite_mean": 0.9964926242828369, "reward_total_composite_std": 0.0008941100095398724, "reward_total_mean": 0.9964926242828369, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9964926242828369, "rewards/meter/std": 0.0008941100095398724, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964926242828369, "rewards/total_composite/std": 0.0008941100095398724, "sampling/importance_sampling_ratio/max": 1.3098853826522827, "sampling/importance_sampling_ratio/mean": 1.0034265518188477, "sampling/importance_sampling_ratio/min": 0.23582126200199127, "sampling/sampling_logp_difference/max": 1.444681167602539, "sampling/sampling_logp_difference/mean": 0.01294564176350832, "step": 1785 }, { "clip_ratio/high_max": 0.004780629184097052, "clip_ratio/high_mean": 0.004780629184097052, "clip_ratio/low_mean": 0.0016556291375309229, "clip_ratio/low_min": 0.0016556291375309229, "clip_ratio/region_mean": 0.0064362583216279745, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 152.125, "completions/mean_terminated_length": 152.125, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.04743872885592282, "epoch": 0.07173555046792786, "frac_reward_zero_std": 0.0, "grad_norm": 2.4926347732543945, "learning_rate": 4.590909090909092e-06, "loss": -0.0127, "num_tokens": 4020442.0, "reward": 0.6688380241394043, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9910025596618652, "reward_meter_std": 0.0011056133080273867, "reward_repeat_penalty_mean": 0.6749999523162842, "reward_repeat_penalty_std": 0.103509820997715, "reward_std": 0.1017415001988411, "reward_total_composite_mean": 0.6688380241394043, "reward_total_composite_std": 0.10174151510000229, "reward_total_mean": 0.6688380241394043, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9910025596618652, "rewards/meter/std": 0.0011056133080273867, "rewards/repeat_penalty/mean": 0.6749999523162842, "rewards/repeat_penalty/std": 0.103509820997715, "rewards/total_composite/mean": 0.6688380241394043, "rewards/total_composite/std": 0.10174151510000229, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0039024353027344, "sampling/importance_sampling_ratio/min": 0.24870102107524872, "sampling/sampling_logp_difference/max": 1.3915038108825684, "sampling/sampling_logp_difference/mean": 0.008964202366769314, "step": 1786 }, { "clip_ratio/high_max": 0.005263741128146648, "clip_ratio/high_mean": 0.005263741128146648, "clip_ratio/low_mean": 0.003947368357330561, "clip_ratio/low_min": 0.003947368357330561, "clip_ratio/region_mean": 0.009211109485477209, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 95.0, "completions/mean_terminated_length": 95.0, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.03921856731176376, "epoch": 0.07177571594971281, "frac_reward_zero_std": 0.0, "grad_norm": 1.5127758979797363, "learning_rate": 4.587878787878788e-06, "loss": 0.0007, "num_tokens": 4022570.0, "reward": 0.9732200503349304, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981739521026611, "reward_meter_std": 6.629144627368078e-05, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07058726251125336, "reward_total_composite_mean": 0.9732200503349304, "reward_total_composite_std": 0.07058726996183395, "reward_total_mean": 0.9732200503349304, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981739521026611, "rewards/meter/std": 6.629144627368078e-05, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9732200503349304, "rewards/total_composite/std": 0.07058726996183395, "sampling/importance_sampling_ratio/max": 1.5315839052200317, "sampling/importance_sampling_ratio/mean": 1.000563383102417, "sampling/importance_sampling_ratio/min": 0.22177138924598694, "sampling/sampling_logp_difference/max": 1.506108283996582, "sampling/sampling_logp_difference/mean": 0.010238762944936752, "step": 1787 }, { "clip_ratio/high_max": 0.038041369058191776, "clip_ratio/high_mean": 0.038041369058191776, "clip_ratio/low_mean": 0.0070457912515848875, "clip_ratio/low_min": 0.0070457912515848875, "clip_ratio/region_mean": 0.045087160309776664, "completions/clipped_ratio": 0.0, "completions/max_length": 198.0, "completions/max_terminated_length": 198.0, "completions/mean_length": 193.875, "completions/mean_terminated_length": 193.875, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.5489388853311539, "epoch": 0.07181588143149777, "frac_reward_zero_std": 0.0, "grad_norm": 2.8437070846557617, "learning_rate": 4.5848484848484854e-06, "loss": 0.0073, "num_tokens": 4025609.0, "reward": 0.9679561853408813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956002235412598, "reward_meter_std": 0.0012398899998515844, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05145645514130592, "reward_total_composite_mean": 0.9679561853408813, "reward_total_composite_std": 0.05145645886659622, "reward_total_mean": 0.9679561853408813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956002235412598, "rewards/meter/std": 0.0012398899998515844, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9679561853408813, "rewards/total_composite/std": 0.05145645886659622, "sampling/importance_sampling_ratio/max": 1.9465110301971436, "sampling/importance_sampling_ratio/mean": 1.008656620979309, "sampling/importance_sampling_ratio/min": 0.2655761241912842, "sampling/sampling_logp_difference/max": 1.3258538246154785, "sampling/sampling_logp_difference/mean": 0.05295930430293083, "step": 1788 }, { "clip_ratio/high_max": 0.016516561503522098, "clip_ratio/high_mean": 0.016516561503522098, "clip_ratio/low_mean": 0.011368778301402926, "clip_ratio/low_min": 0.011368778301402926, "clip_ratio/region_mean": 0.027885339804925025, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 67.625, "completions/mean_terminated_length": 67.625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.38035739958286285, "epoch": 0.07185604691328272, "frac_reward_zero_std": 0.0, "grad_norm": 3.5395383834838867, "learning_rate": 4.581818181818183e-06, "loss": 0.0114, "num_tokens": 4027414.0, "reward": 0.9397947192192078, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9397947192192078, "reward_meter_std": 0.026229292154312134, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.026229292154312134, "reward_total_composite_mean": 0.9397947192192078, "reward_total_composite_std": 0.026229292154312134, "reward_total_mean": 0.9397947192192078, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9397947192192078, "rewards/meter/std": 0.026229292154312134, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9397947192192078, "rewards/total_composite/std": 0.026229292154312134, "sampling/importance_sampling_ratio/max": 1.5871034860610962, "sampling/importance_sampling_ratio/mean": 1.0007556676864624, "sampling/importance_sampling_ratio/min": 0.3197151720523834, "sampling/sampling_logp_difference/max": 1.1403248310089111, "sampling/sampling_logp_difference/mean": 0.04796122387051582, "step": 1789 }, { "clip_ratio/high_max": 0.013679954456165433, "clip_ratio/high_mean": 0.013679954456165433, "clip_ratio/low_mean": 0.019280804554000497, "clip_ratio/low_min": 0.019280804554000497, "clip_ratio/region_mean": 0.03296075901016593, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 110.625, "completions/mean_terminated_length": 110.625, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.4972764067351818, "epoch": 0.07189621239506767, "frac_reward_zero_std": 0.0, "grad_norm": 2.5047359466552734, "learning_rate": 4.578787878787879e-06, "loss": 0.0094, "num_tokens": 4029603.0, "reward": 0.9982960224151611, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982960224151611, "reward_meter_std": 0.000550406810361892, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005504076834768057, "reward_total_composite_mean": 0.9982960224151611, "reward_total_composite_std": 0.000550406810361892, "reward_total_mean": 0.9982960224151611, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982960224151611, "rewards/meter/std": 0.000550406810361892, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982960224151611, "rewards/total_composite/std": 0.000550406810361892, "sampling/importance_sampling_ratio/max": 1.5884928703308105, "sampling/importance_sampling_ratio/mean": 1.005360722541809, "sampling/importance_sampling_ratio/min": 0.35043227672576904, "sampling/sampling_logp_difference/max": 1.0485877990722656, "sampling/sampling_logp_difference/mean": 0.048308465629816055, "step": 1790 }, { "clip_ratio/high_max": 0.015547977527603507, "clip_ratio/high_mean": 0.015547977527603507, "clip_ratio/low_mean": 0.009515275072772056, "clip_ratio/low_min": 0.009515275072772056, "clip_ratio/region_mean": 0.025063252600375563, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 280.75, "completions/mean_terminated_length": 280.75, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.39633841440081596, "epoch": 0.07193637787685263, "frac_reward_zero_std": 0.0, "grad_norm": 2.072072982788086, "learning_rate": 4.575757575757576e-06, "loss": -0.0086, "num_tokens": 4033409.0, "reward": 0.7775777578353882, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.921875, "reward_count_adherence_std": 0.06469365209341049, "reward_meter_mean": 0.9915696978569031, "reward_meter_std": 0.01817433536052704, "reward_repeat_penalty_mean": 0.8535714149475098, "reward_repeat_penalty_std": 0.07463330775499344, "reward_std": 0.05556074529886246, "reward_total_composite_mean": 0.7775777578353882, "reward_total_composite_std": 0.05556074157357216, "reward_total_mean": 0.7775777578353882, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.921875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/meter/mean": 0.9915696978569031, "rewards/meter/std": 0.01817433536052704, "rewards/repeat_penalty/mean": 0.8535714149475098, "rewards/repeat_penalty/std": 0.07463330775499344, "rewards/total_composite/mean": 0.7775777578353882, "rewards/total_composite/std": 0.05556074157357216, "sampling/importance_sampling_ratio/max": 1.9100106954574585, "sampling/importance_sampling_ratio/mean": 1.0069139003753662, "sampling/importance_sampling_ratio/min": 0.14663827419281006, "sampling/sampling_logp_difference/max": 1.9197864532470703, "sampling/sampling_logp_difference/mean": 0.03709190711379051, "step": 1791 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.04893740452826023, "epoch": 0.07197654335863758, "frac_reward_zero_std": 0.0, "grad_norm": 0.36933040618896484, "learning_rate": 4.572727272727273e-06, "loss": 0.0012, "num_tokens": 4035226.0, "reward": 0.9975280165672302, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975280165672302, "reward_meter_std": 4.185540819889866e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.18471208831761e-05, "reward_total_composite_mean": 0.9975280165672302, "reward_total_composite_std": 4.185540819889866e-05, "reward_total_mean": 0.9975280165672302, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975280165672302, "rewards/meter/std": 4.185540819889866e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975280165672302, "rewards/total_composite/std": 4.185540819889866e-05, "sampling/importance_sampling_ratio/max": 1.4639054536819458, "sampling/importance_sampling_ratio/mean": 1.0018044710159302, "sampling/importance_sampling_ratio/min": 0.516437292098999, "sampling/sampling_logp_difference/max": 0.6608014106750488, "sampling/sampling_logp_difference/mean": 0.0064887660555541515, "step": 1792 }, { "clip_ratio/high_max": 0.0013157895300537348, "clip_ratio/high_mean": 0.0013157895300537348, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0013157895300537348, "completions/clipped_ratio": 0.0, "completions/max_length": 95.0, "completions/max_terminated_length": 95.0, "completions/mean_length": 94.625, "completions/mean_terminated_length": 94.625, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.021367451641708612, "epoch": 0.07201670884042254, "frac_reward_zero_std": 0.0, "grad_norm": 2.3365345001220703, "learning_rate": 4.56969696969697e-06, "loss": 0.0024, "num_tokens": 4037239.0, "reward": 0.8733782768249512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981467723846436, "reward_meter_std": 4.623903805622831e-05, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10331728309392929, "reward_total_composite_mean": 0.8733782768249512, "reward_total_composite_std": 0.1033172756433487, "reward_total_mean": 0.8733782768249512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981467723846436, "rewards/meter/std": 4.623903805622831e-05, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8733782768249512, "rewards/total_composite/std": 0.1033172756433487, "sampling/importance_sampling_ratio/max": 1.2057961225509644, "sampling/importance_sampling_ratio/mean": 1.0002309083938599, "sampling/importance_sampling_ratio/min": 0.5343936085700989, "sampling/sampling_logp_difference/max": 0.6266226768493652, "sampling/sampling_logp_difference/mean": 0.003089956007897854, "step": 1793 }, { "clip_ratio/high_max": 0.002393221133388579, "clip_ratio/high_mean": 0.002393221133388579, "clip_ratio/low_mean": 0.0030517695122398436, "clip_ratio/low_min": 0.0030517695122398436, "clip_ratio/region_mean": 0.0054449906456284225, "completions/clipped_ratio": 0.0, "completions/max_length": 213.0, "completions/max_terminated_length": 213.0, "completions/mean_length": 205.875, "completions/mean_terminated_length": 205.875, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "entropy": 0.0551890037022531, "epoch": 0.07205687432220749, "frac_reward_zero_std": 0.0, "grad_norm": 1.2437056303024292, "learning_rate": 4.566666666666667e-06, "loss": -0.0075, "num_tokens": 4040774.0, "reward": 0.6546941995620728, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9908320307731628, "reward_meter_std": 0.000867572205606848, "reward_repeat_penalty_mean": 0.6607142686843872, "reward_repeat_penalty_std": 0.06331465393304825, "reward_std": 0.06324288249015808, "reward_total_composite_mean": 0.6546941995620728, "reward_total_composite_std": 0.06324289739131927, "reward_total_mean": 0.6546941995620728, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9908320307731628, "rewards/meter/std": 0.000867572205606848, "rewards/repeat_penalty/mean": 0.6607142686843872, "rewards/repeat_penalty/std": 0.06331465393304825, "rewards/total_composite/mean": 0.6546941995620728, "rewards/total_composite/std": 0.06324289739131927, "sampling/importance_sampling_ratio/max": 1.7237244844436646, "sampling/importance_sampling_ratio/mean": 1.0008214712142944, "sampling/importance_sampling_ratio/min": 0.26433053612709045, "sampling/sampling_logp_difference/max": 1.3305549621582031, "sampling/sampling_logp_difference/mean": 0.009390618652105331, "step": 1794 }, { "clip_ratio/high_max": 0.030286709661595523, "clip_ratio/high_mean": 0.030286709661595523, "clip_ratio/low_mean": 0.004201680887490511, "clip_ratio/low_min": 0.004201680887490511, "clip_ratio/region_mean": 0.034488390549086034, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 118.5, "completions/mean_terminated_length": 118.5, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.46814702078700066, "epoch": 0.07209703980399244, "frac_reward_zero_std": 0.0, "grad_norm": 2.7669808864593506, "learning_rate": 4.563636363636364e-06, "loss": 0.0056, "num_tokens": 4043010.0, "reward": 0.9722900390625, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971722364425659, "reward_meter_std": 0.0011399141512811184, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07114371657371521, "reward_total_composite_mean": 0.9722900390625, "reward_total_composite_std": 0.0711437240242958, "reward_total_mean": 0.9722900390625, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971722364425659, "rewards/meter/std": 0.0011399141512811184, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9722900390625, "rewards/total_composite/std": 0.0711437240242958, "sampling/importance_sampling_ratio/max": 1.7191333770751953, "sampling/importance_sampling_ratio/mean": 1.014491319656372, "sampling/importance_sampling_ratio/min": 0.28391486406326294, "sampling/sampling_logp_difference/max": 1.2590808868408203, "sampling/sampling_logp_difference/mean": 0.04540203511714935, "step": 1795 }, { "clip_ratio/high_max": 0.024542337749153376, "clip_ratio/high_mean": 0.024542337749153376, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/region_mean": 0.027201912133023143, "completions/clipped_ratio": 0.0, "completions/max_length": 199.0, "completions/max_terminated_length": 199.0, "completions/mean_length": 184.75, "completions/mean_terminated_length": 184.75, "completions/min_length": 168.0, "completions/min_terminated_length": 168.0, "entropy": 0.5114907324314117, "epoch": 0.0721372052857774, "frac_reward_zero_std": 0.0, "grad_norm": 2.253188371658325, "learning_rate": 4.560606060606061e-06, "loss": 0.0069, "num_tokens": 4045960.0, "reward": 0.7317129969596863, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.9870539903640747, "reward_meter_std": 0.024071279913187027, "reward_repeat_penalty_mean": 0.8888888955116272, "reward_repeat_penalty_std": 0.10286889225244522, "reward_std": 0.30691882967948914, "reward_total_composite_mean": 0.7317129969596863, "reward_total_composite_std": 0.30691879987716675, "reward_total_mean": 0.7317129969596863, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.9870539903640747, "rewards/meter/std": 0.024071279913187027, "rewards/repeat_penalty/mean": 0.8888888955116272, "rewards/repeat_penalty/std": 0.10286889225244522, "rewards/total_composite/mean": 0.7317129969596863, "rewards/total_composite/std": 0.30691879987716675, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0125421285629272, "sampling/importance_sampling_ratio/min": 0.27392005920410156, "sampling/sampling_logp_difference/max": 1.2949190139770508, "sampling/sampling_logp_difference/mean": 0.05264398828148842, "step": 1796 }, { "clip_ratio/high_max": 0.01244131033308804, "clip_ratio/high_mean": 0.01244131033308804, "clip_ratio/low_mean": 0.010334618214983493, "clip_ratio/low_min": 0.010334618214983493, "clip_ratio/region_mean": 0.022775928548071533, "completions/clipped_ratio": 0.0, "completions/max_length": 458.0, "completions/max_terminated_length": 458.0, "completions/mean_length": 420.375, "completions/mean_terminated_length": 420.375, "completions/min_length": 395.0, "completions/min_terminated_length": 395.0, "entropy": 0.22779023833572865, "epoch": 0.07217737076756235, "frac_reward_zero_std": 0.0, "grad_norm": 1.9448902606964111, "learning_rate": 4.557575757575758e-06, "loss": 0.0079, "num_tokens": 4051243.0, "reward": 0.2035287767648697, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.023145509883761406, "reward_meter_mean": 0.448710560798645, "reward_meter_std": 0.4174151122570038, "reward_repeat_penalty_mean": 0.7848790884017944, "reward_repeat_penalty_std": 0.13375473022460938, "reward_std": 0.17783081531524658, "reward_total_composite_mean": 0.2035287767648697, "reward_total_composite_std": 0.17783083021640778, "reward_total_mean": 0.2035287767648697, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.023145509883761406, "rewards/meter/mean": 0.448710560798645, "rewards/meter/std": 0.4174151122570038, "rewards/repeat_penalty/mean": 0.7848790884017944, "rewards/repeat_penalty/std": 0.13375473022460938, "rewards/total_composite/mean": 0.2035287767648697, "rewards/total_composite/std": 0.17783083021640778, "sampling/importance_sampling_ratio/max": 1.8132941722869873, "sampling/importance_sampling_ratio/mean": 1.0021564960479736, "sampling/importance_sampling_ratio/min": 0.0003370313497725874, "sampling/sampling_logp_difference/max": 7.995334625244141, "sampling/sampling_logp_difference/mean": 0.036439694464206696, "step": 1797 }, { "clip_ratio/high_max": 0.017403612146154046, "clip_ratio/high_mean": 0.017403612146154046, "clip_ratio/low_mean": 0.013545665598940104, "clip_ratio/low_min": 0.013545665598940104, "clip_ratio/region_mean": 0.03094927774509415, "completions/clipped_ratio": 0.0, "completions/max_length": 188.0, "completions/max_terminated_length": 188.0, "completions/mean_length": 182.5, "completions/mean_terminated_length": 182.5, "completions/min_length": 172.0, "completions/min_terminated_length": 172.0, "entropy": 0.3780089020729065, "epoch": 0.0722175362493473, "frac_reward_zero_std": 0.0, "grad_norm": 2.2561349868774414, "learning_rate": 4.554545454545455e-06, "loss": 0.0113, "num_tokens": 4054231.0, "reward": 0.9284070730209351, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977061152458191, "reward_meter_std": 0.001272720517590642, "reward_repeat_penalty_mean": 0.9305555820465088, "reward_repeat_penalty_std": 0.08266931027173996, "reward_std": 0.08227662742137909, "reward_total_composite_mean": 0.9284070730209351, "reward_total_composite_std": 0.0822766125202179, "reward_total_mean": 0.9284070730209351, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977061152458191, "rewards/meter/std": 0.001272720517590642, "rewards/repeat_penalty/mean": 0.9305555820465088, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.9284070730209351, "rewards/total_composite/std": 0.0822766125202179, "sampling/importance_sampling_ratio/max": 1.7942215204238892, "sampling/importance_sampling_ratio/mean": 1.008242130279541, "sampling/importance_sampling_ratio/min": 0.15907828509807587, "sampling/sampling_logp_difference/max": 1.8383588790893555, "sampling/sampling_logp_difference/mean": 0.03487614914774895, "step": 1798 }, { "clip_ratio/high_max": 0.019246643991209567, "clip_ratio/high_mean": 0.019246643991209567, "clip_ratio/low_mean": 0.018777355086058378, "clip_ratio/low_min": 0.018777355086058378, "clip_ratio/region_mean": 0.038023999077267945, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 68.75, "completions/mean_terminated_length": 68.75, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.5148973725736141, "epoch": 0.07225770173113226, "frac_reward_zero_std": 0.0, "grad_norm": 5.813068866729736, "learning_rate": 4.551515151515152e-06, "loss": -0.0358, "num_tokens": 4056037.0, "reward": 0.7557946443557739, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7557946443557739, "reward_meter_std": 0.27249956130981445, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.27249956130981445, "reward_total_composite_mean": 0.7557946443557739, "reward_total_composite_std": 0.27249956130981445, "reward_total_mean": 0.7557946443557739, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7557946443557739, "rewards/meter/std": 0.27249956130981445, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7557946443557739, "rewards/total_composite/std": 0.27249956130981445, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0174649953842163, "sampling/importance_sampling_ratio/min": 0.40181437134742737, "sampling/sampling_logp_difference/max": 0.9117650985717773, "sampling/sampling_logp_difference/mean": 0.054108865559101105, "step": 1799 }, { "clip_ratio/high_max": 0.05649038520641625, "clip_ratio/high_mean": 0.05649038520641625, "clip_ratio/low_mean": 0.01666666753590107, "clip_ratio/low_min": 0.01666666753590107, "clip_ratio/region_mean": 0.07315705274231732, "completions/clipped_ratio": 0.0, "completions/max_length": 45.0, "completions/max_terminated_length": 45.0, "completions/mean_length": 40.375, "completions/mean_terminated_length": 40.375, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.5478580147027969, "epoch": 0.07229786721291721, "frac_reward_zero_std": 0.0, "grad_norm": 13.604365348815918, "learning_rate": 4.548484848484849e-06, "loss": 0.0696, "num_tokens": 4057576.0, "reward": 0.9156763553619385, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9156763553619385, "reward_meter_std": 0.20896102488040924, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.20896100997924805, "reward_total_composite_mean": 0.9156763553619385, "reward_total_composite_std": 0.20896102488040924, "reward_total_mean": 0.9156763553619385, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9156763553619385, "rewards/meter/std": 0.20896102488040924, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9156763553619385, "rewards/total_composite/std": 0.20896102488040924, "sampling/importance_sampling_ratio/max": 1.8849577903747559, "sampling/importance_sampling_ratio/mean": 1.0134040117263794, "sampling/importance_sampling_ratio/min": 0.217363640666008, "sampling/sampling_logp_difference/max": 1.5261836051940918, "sampling/sampling_logp_difference/mean": 0.07238361239433289, "step": 1800 }, { "epoch": 0.07229786721291721, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 341.0, "eval_completions/max_terminated_length": 341.0, "eval_completions/mean_length": 192.78846153846155, "eval_completions/mean_terminated_length": 192.78846153846155, "eval_completions/min_length": 63.84615384615385, "eval_completions/min_terminated_length": 63.84615384615385, "eval_entropy": 0.3313331672778496, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 4057576.0, "eval_reward": 0.5133193800082574, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.8753787233279302, "eval_reward_count_adherence_std": 0.13936235010623932, "eval_reward_meter_mean": 0.6883980471354264, "eval_reward_meter_std": 0.388204580746018, "eval_reward_repeat_penalty_mean": 0.8605746856102576, "eval_reward_repeat_penalty_std": 0.13569565231983477, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5133193800082574, "eval_reward_total_composite_std": 0.346407216328841, "eval_reward_total_mean": 0.5133193800082574, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.8753787233279302, "eval_rewards/count_adherence/std": 0.13936235010623932, "eval_rewards/meter/mean": 0.6883980471354264, "eval_rewards/meter/std": 0.388204580746018, "eval_rewards/repeat_penalty/mean": 0.8605746856102576, "eval_rewards/repeat_penalty/std": 0.13569565231983477, "eval_rewards/total_composite/mean": 0.5133193800082574, "eval_rewards/total_composite/std": 0.346407216328841, "eval_runtime": 66.0173, "eval_samples_per_second": 1.575, "eval_sampling/importance_sampling_ratio/max": 1.5431275459436269, "eval_sampling/importance_sampling_ratio/mean": 1.0084307743952825, "eval_sampling/importance_sampling_ratio/min": 0.36259872638262236, "eval_sampling/sampling_logp_difference/max": 1.0275482031015248, "eval_sampling/sampling_logp_difference/mean": 0.02931356322593414, "eval_steps_per_second": 0.197, "step": 1800 }, { "clip_ratio/high_max": 0.027773853624239564, "clip_ratio/high_mean": 0.027773853624239564, "clip_ratio/low_mean": 0.02372721955180168, "clip_ratio/low_min": 0.02372721955180168, "clip_ratio/region_mean": 0.051501073176041245, "completions/clipped_ratio": 0.0, "completions/max_length": 254.0, "completions/max_terminated_length": 254.0, "completions/mean_length": 241.625, "completions/mean_terminated_length": 241.625, "completions/min_length": 227.0, "completions/min_terminated_length": 227.0, "entropy": 0.6957518979907036, "epoch": 0.07233803269470217, "frac_reward_zero_std": 0.0, "grad_norm": 4.170017719268799, "learning_rate": 4.5454545454545455e-06, "loss": 0.0329, "num_tokens": 4061109.0, "reward": 0.39037013053894043, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5351053476333618, "reward_meter_std": 0.38701698184013367, "reward_repeat_penalty_mean": 0.9261363744735718, "reward_repeat_penalty_std": 0.05368336662650108, "reward_std": 0.3217414319515228, "reward_total_composite_mean": 0.39037013053894043, "reward_total_composite_std": 0.3217414319515228, "reward_total_mean": 0.39037013053894043, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5351053476333618, "rewards/meter/std": 0.38701698184013367, "rewards/repeat_penalty/mean": 0.9261363744735718, "rewards/repeat_penalty/std": 0.05368336662650108, "rewards/total_composite/mean": 0.39037013053894043, "rewards/total_composite/std": 0.3217414319515228, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0214029550552368, "sampling/importance_sampling_ratio/min": 0.07254444062709808, "sampling/sampling_logp_difference/max": 2.623555898666382, "sampling/sampling_logp_difference/mean": 0.0780494287610054, "step": 1801 }, { "clip_ratio/high_max": 0.01266491471324116, "clip_ratio/high_mean": 0.01266491471324116, "clip_ratio/low_mean": 0.017737353453412652, "clip_ratio/low_min": 0.017737353453412652, "clip_ratio/region_mean": 0.030402268166653812, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 103.0, "completions/mean_terminated_length": 103.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.3216414675116539, "epoch": 0.07237819817648712, "frac_reward_zero_std": 0.0, "grad_norm": 4.134006500244141, "learning_rate": 4.542424242424243e-06, "loss": 0.038, "num_tokens": 4063157.0, "reward": 0.8030074238777161, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9173542261123657, "reward_meter_std": 0.045704178512096405, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10497336089611053, "reward_total_composite_mean": 0.8030074238777161, "reward_total_composite_std": 0.10497334599494934, "reward_total_mean": 0.8030074238777161, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9173542261123657, "rewards/meter/std": 0.045704178512096405, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.8030074238777161, "rewards/total_composite/std": 0.10497334599494934, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0087733268737793, "sampling/importance_sampling_ratio/min": 0.18847519159317017, "sampling/sampling_logp_difference/max": 1.6687889099121094, "sampling/sampling_logp_difference/mean": 0.036168623715639114, "step": 1802 }, { "clip_ratio/high_max": 0.02186147216707468, "clip_ratio/high_mean": 0.02186147216707468, "clip_ratio/low_mean": 0.022176598431542516, "clip_ratio/low_min": 0.022176598431542516, "clip_ratio/region_mean": 0.044038070598617196, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.5, "completions/mean_terminated_length": 34.5, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.6431399546563625, "epoch": 0.07241836365827208, "frac_reward_zero_std": 0.0, "grad_norm": 7.020390510559082, "learning_rate": 4.539393939393939e-06, "loss": -0.0189, "num_tokens": 4064817.0, "reward": 0.9750309586524963, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9750309586524963, "reward_meter_std": 0.01676092855632305, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01676092855632305, "reward_total_composite_mean": 0.9750309586524963, "reward_total_composite_std": 0.01676092855632305, "reward_total_mean": 0.9750309586524963, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9750309586524963, "rewards/meter/std": 0.01676092855632305, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9750309586524963, "rewards/total_composite/std": 0.01676092855632305, "sampling/importance_sampling_ratio/max": 1.694462537765503, "sampling/importance_sampling_ratio/mean": 1.008634090423584, "sampling/importance_sampling_ratio/min": 0.2331681251525879, "sampling/sampling_logp_difference/max": 1.4559955596923828, "sampling/sampling_logp_difference/mean": 0.06769955903291702, "step": 1803 }, { "clip_ratio/high_max": 0.010927102295681834, "clip_ratio/high_mean": 0.010927102295681834, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.010927102295681834, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 57.25, "completions/mean_terminated_length": 57.25, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.13894780911505222, "epoch": 0.07245852914005703, "frac_reward_zero_std": 0.0, "grad_norm": 2.0322115421295166, "learning_rate": 4.5363636363636364e-06, "loss": 0.0014, "num_tokens": 4066459.0, "reward": 0.9932329654693604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9932329654693604, "reward_meter_std": 0.0008318190230056643, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008318254840560257, "reward_total_composite_mean": 0.9932329654693604, "reward_total_composite_std": 0.0008318190230056643, "reward_total_mean": 0.9932329654693604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9932329654693604, "rewards/meter/std": 0.0008318190230056643, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9932329654693604, "rewards/total_composite/std": 0.0008318190230056643, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0055011510849, "sampling/importance_sampling_ratio/min": 0.4126149117946625, "sampling/sampling_logp_difference/max": 1.751237392425537, "sampling/sampling_logp_difference/mean": 0.026665696874260902, "step": 1804 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/region_mean": 0.0019841270986944437, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.09062223881483078, "epoch": 0.07249869462184198, "frac_reward_zero_std": 0.0, "grad_norm": 2.5561840534210205, "learning_rate": 4.533333333333334e-06, "loss": -0.0002, "num_tokens": 4068131.0, "reward": 0.9917310476303101, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9917310476303101, "reward_meter_std": 0.014378287829458714, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014378308318555355, "reward_total_composite_mean": 0.9917310476303101, "reward_total_composite_std": 0.014378287829458714, "reward_total_mean": 0.9917310476303101, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9917310476303101, "rewards/meter/std": 0.014378287829458714, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917310476303101, "rewards/total_composite/std": 0.014378287829458714, "sampling/importance_sampling_ratio/max": 1.4607477188110352, "sampling/importance_sampling_ratio/mean": 1.0059313774108887, "sampling/importance_sampling_ratio/min": 0.612943172454834, "sampling/sampling_logp_difference/max": 0.489483118057251, "sampling/sampling_logp_difference/mean": 0.010517938062548637, "step": 1805 }, { "clip_ratio/high_max": 0.03447378391865641, "clip_ratio/high_mean": 0.03447378391865641, "clip_ratio/low_mean": 0.013943745056167245, "clip_ratio/low_min": 0.013943745056167245, "clip_ratio/region_mean": 0.048417528974823654, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 80.375, "completions/mean_terminated_length": 80.375, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.9013412185013294, "epoch": 0.07253886010362694, "frac_reward_zero_std": 0.0, "grad_norm": 5.9204182624816895, "learning_rate": 4.53030303030303e-06, "loss": 0.0586, "num_tokens": 4069918.0, "reward": 0.6509301066398621, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.774915874004364, "reward_meter_std": 0.3431837856769562, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.42339858412742615, "reward_total_composite_mean": 0.6509301066398621, "reward_total_composite_std": 0.42339858412742615, "reward_total_mean": 0.6509301066398621, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.774915874004364, "rewards/meter/std": 0.3431837856769562, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6509301066398621, "rewards/total_composite/std": 0.42339858412742615, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.022961139678955, "sampling/importance_sampling_ratio/min": 0.02298472262918949, "sampling/sampling_logp_difference/max": 3.772925615310669, "sampling/sampling_logp_difference/mean": 0.09225737303495407, "step": 1806 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.008621971355751157, "clip_ratio/low_min": 0.008621971355751157, "clip_ratio/region_mean": 0.008621971355751157, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 57.375, "completions/mean_terminated_length": 57.375, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.13474871311336756, "epoch": 0.07257902558541189, "frac_reward_zero_std": 0.0, "grad_norm": 3.1104016304016113, "learning_rate": 4.527272727272727e-06, "loss": 0.0059, "num_tokens": 4071641.0, "reward": 0.9939553737640381, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939553737640381, "reward_meter_std": 0.0008149455534294248, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008149410132318735, "reward_total_composite_mean": 0.9939553737640381, "reward_total_composite_std": 0.0008149455534294248, "reward_total_mean": 0.9939553737640381, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939553737640381, "rewards/meter/std": 0.0008149455534294248, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9939553737640381, "rewards/total_composite/std": 0.0008149455534294248, "sampling/importance_sampling_ratio/max": 1.5543711185455322, "sampling/importance_sampling_ratio/mean": 1.0032254457473755, "sampling/importance_sampling_ratio/min": 0.41036123037338257, "sampling/sampling_logp_difference/max": 0.8907175064086914, "sampling/sampling_logp_difference/mean": 0.021058987826108932, "step": 1807 }, { "clip_ratio/high_max": 0.039301327895373106, "clip_ratio/high_mean": 0.039301327895373106, "clip_ratio/low_mean": 0.01045380299910903, "clip_ratio/low_min": 0.01045380299910903, "clip_ratio/region_mean": 0.049755130894482136, "completions/clipped_ratio": 0.0, "completions/max_length": 169.0, "completions/max_terminated_length": 169.0, "completions/mean_length": 157.625, "completions/mean_terminated_length": 157.625, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.9014198370277882, "epoch": 0.07261919106719684, "frac_reward_zero_std": 0.0, "grad_norm": 4.073712348937988, "learning_rate": 4.524242424242425e-06, "loss": -0.0419, "num_tokens": 4074374.0, "reward": 0.7510942220687866, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.8494125604629517, "reward_meter_std": 0.2693995535373688, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.41580459475517273, "reward_total_composite_mean": 0.7510942220687866, "reward_total_composite_std": 0.41580459475517273, "reward_total_mean": 0.7510942220687866, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.8494125604629517, "rewards/meter/std": 0.2693995535373688, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7510942220687866, "rewards/total_composite/std": 0.41580459475517273, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0168873071670532, "sampling/importance_sampling_ratio/min": 0.15204578638076782, "sampling/sampling_logp_difference/max": 1.8835735321044922, "sampling/sampling_logp_difference/mean": 0.0892598107457161, "step": 1808 }, { "clip_ratio/high_max": 0.044713106006383896, "clip_ratio/high_mean": 0.044713106006383896, "clip_ratio/low_mean": 0.017330652568489313, "clip_ratio/low_min": 0.017330652568489313, "clip_ratio/region_mean": 0.06204375857487321, "completions/clipped_ratio": 0.0, "completions/max_length": 172.0, "completions/max_terminated_length": 172.0, "completions/mean_length": 160.5, "completions/mean_terminated_length": 160.5, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "entropy": 1.1915706768631935, "epoch": 0.0726593565489818, "frac_reward_zero_std": 0.0, "grad_norm": 4.39199161529541, "learning_rate": 4.521212121212122e-06, "loss": 0.0062, "num_tokens": 4077130.0, "reward": 0.9022413492202759, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9320476055145264, "reward_meter_std": 0.11087671667337418, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.13386201858520508, "reward_total_composite_mean": 0.9022413492202759, "reward_total_composite_std": 0.13386203348636627, "reward_total_mean": 0.9022413492202759, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9320476055145264, "rewards/meter/std": 0.11087671667337418, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9022413492202759, "rewards/total_composite/std": 0.13386203348636627, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0246384143829346, "sampling/importance_sampling_ratio/min": 0.21437355875968933, "sampling/sampling_logp_difference/max": 1.5400352478027344, "sampling/sampling_logp_difference/mean": 0.10231172293424606, "step": 1809 }, { "clip_ratio/high_max": 0.012699689657893032, "clip_ratio/high_mean": 0.012699689657893032, "clip_ratio/low_mean": 0.006453010253608227, "clip_ratio/low_min": 0.006453010253608227, "clip_ratio/region_mean": 0.01915269991150126, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 136.0, "completions/mean_terminated_length": 136.0, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.24091620557010174, "epoch": 0.07269952203076675, "frac_reward_zero_std": 0.0, "grad_norm": 2.1702988147735596, "learning_rate": 4.518181818181819e-06, "loss": -0.0001, "num_tokens": 4079618.0, "reward": 0.7060449123382568, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8782405853271484, "reward_meter_std": 0.031040111556649208, "reward_repeat_penalty_mean": 0.8035714030265808, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07351230829954147, "reward_total_composite_mean": 0.7060449123382568, "reward_total_composite_std": 0.07351230084896088, "reward_total_mean": 0.7060449123382568, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8782405853271484, "rewards/meter/std": 0.031040111556649208, "rewards/repeat_penalty/mean": 0.8035714030265808, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.7060449123382568, "rewards/total_composite/std": 0.07351230084896088, "sampling/importance_sampling_ratio/max": 1.6531912088394165, "sampling/importance_sampling_ratio/mean": 1.0071299076080322, "sampling/importance_sampling_ratio/min": 0.30320504307746887, "sampling/sampling_logp_difference/max": 1.1933460235595703, "sampling/sampling_logp_difference/mean": 0.022378606721758842, "step": 1810 }, { "clip_ratio/high_max": 0.037365528754889965, "clip_ratio/high_mean": 0.037365528754889965, "clip_ratio/low_mean": 0.006021729204803705, "clip_ratio/low_min": 0.006021729204803705, "clip_ratio/region_mean": 0.04338725795969367, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 104.375, "completions/mean_terminated_length": 104.375, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.5101095288991928, "epoch": 0.0727396875125517, "frac_reward_zero_std": 0.0, "grad_norm": 4.185992240905762, "learning_rate": 4.5151515151515155e-06, "loss": 0.0067, "num_tokens": 4081773.0, "reward": 0.8305845260620117, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8305845260620117, "reward_meter_std": 0.25914400815963745, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.25914397835731506, "reward_total_composite_mean": 0.8305845260620117, "reward_total_composite_std": 0.25914400815963745, "reward_total_mean": 0.8305845260620117, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8305845260620117, "rewards/meter/std": 0.25914400815963745, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8305845260620117, "rewards/total_composite/std": 0.25914400815963745, "sampling/importance_sampling_ratio/max": 1.7020655870437622, "sampling/importance_sampling_ratio/mean": 1.0125762224197388, "sampling/importance_sampling_ratio/min": 0.21204298734664917, "sampling/sampling_logp_difference/max": 1.5509662628173828, "sampling/sampling_logp_difference/mean": 0.058025941252708435, "step": 1811 }, { "clip_ratio/high_max": 0.009920635493472219, "clip_ratio/high_mean": 0.009920635493472219, "clip_ratio/low_mean": 0.003937252098694444, "clip_ratio/low_min": 0.003937252098694444, "clip_ratio/region_mean": 0.013857887592166662, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.625, "completions/mean_terminated_length": 62.625, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.13391203712671995, "epoch": 0.07277985299433666, "frac_reward_zero_std": 0.0, "grad_norm": 7.030357360839844, "learning_rate": 4.512121212121213e-06, "loss": 0.0156, "num_tokens": 4083458.0, "reward": 0.7619152069091797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7619152069091797, "reward_meter_std": 0.4342895746231079, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4342895746231079, "reward_total_composite_mean": 0.7619152069091797, "reward_total_composite_std": 0.4342895746231079, "reward_total_mean": 0.7619152069091797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7619152069091797, "rewards/meter/std": 0.4342895746231079, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7619152069091797, "rewards/total_composite/std": 0.4342895746231079, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0049937963485718, "sampling/importance_sampling_ratio/min": 0.3256944417953491, "sampling/sampling_logp_difference/max": 1.121795654296875, "sampling/sampling_logp_difference/mean": 0.024434061720967293, "step": 1812 }, { "clip_ratio/high_max": 0.015669515822082758, "clip_ratio/high_mean": 0.015669515822082758, "clip_ratio/low_mean": 0.030510480049997568, "clip_ratio/low_min": 0.030510480049997568, "clip_ratio/region_mean": 0.046179995872080326, "completions/clipped_ratio": 0.0, "completions/max_length": 195.0, "completions/max_terminated_length": 195.0, "completions/mean_length": 176.5, "completions/mean_terminated_length": 176.5, "completions/min_length": 160.0, "completions/min_terminated_length": 160.0, "entropy": 0.7889967784285545, "epoch": 0.07282001847612161, "frac_reward_zero_std": 0.0, "grad_norm": 3.172938346862793, "learning_rate": 4.50909090909091e-06, "loss": -0.0181, "num_tokens": 4086534.0, "reward": 0.8140916228294373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8500000238418579, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.9555693864822388, "reward_meter_std": 0.09552352130413055, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.132888063788414, "reward_total_composite_mean": 0.8140916228294373, "reward_total_composite_std": 0.132888063788414, "reward_total_mean": 0.8140916228294373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8500000238418579, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.9555693864822388, "rewards/meter/std": 0.09552352130413055, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8140916228294373, "rewards/total_composite/std": 0.132888063788414, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0195351839065552, "sampling/importance_sampling_ratio/min": 0.2596273720264435, "sampling/sampling_logp_difference/max": 1.3485078811645508, "sampling/sampling_logp_difference/mean": 0.07848730683326721, "step": 1813 }, { "clip_ratio/high_max": 0.0025775935500860214, "clip_ratio/high_mean": 0.0025775935500860214, "clip_ratio/low_mean": 0.0013440860202535987, "clip_ratio/low_min": 0.0013440860202535987, "clip_ratio/region_mean": 0.00392167957033962, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 95.75, "completions/mean_terminated_length": 95.75, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.04789540730416775, "epoch": 0.07286018395790657, "frac_reward_zero_std": 0.0, "grad_norm": 2.741095781326294, "learning_rate": 4.5060606060606065e-06, "loss": -0.0068, "num_tokens": 4088604.0, "reward": 0.9976317882537842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976317882537842, "reward_meter_std": 0.0016386967618018389, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0016386950155720115, "reward_total_composite_mean": 0.9976317882537842, "reward_total_composite_std": 0.0016386967618018389, "reward_total_mean": 0.9976317882537842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976317882537842, "rewards/meter/std": 0.0016386967618018389, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976317882537842, "rewards/total_composite/std": 0.0016386967618018389, "sampling/importance_sampling_ratio/max": 1.3251795768737793, "sampling/importance_sampling_ratio/mean": 1.0037214756011963, "sampling/importance_sampling_ratio/min": 0.8114128112792969, "sampling/sampling_logp_difference/max": 0.28154802322387695, "sampling/sampling_logp_difference/mean": 0.006327650509774685, "step": 1814 }, { "clip_ratio/high_max": 0.004235348082147539, "clip_ratio/high_mean": 0.004235348082147539, "clip_ratio/low_mean": 0.005773193319328129, "clip_ratio/low_min": 0.005773193319328129, "clip_ratio/region_mean": 0.010008541401475668, "completions/clipped_ratio": 0.0, "completions/max_length": 185.0, "completions/max_terminated_length": 185.0, "completions/mean_length": 175.0, "completions/mean_terminated_length": 175.0, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.12415077816694975, "epoch": 0.07290034943969152, "frac_reward_zero_std": 0.0, "grad_norm": 2.079590320587158, "learning_rate": 4.503030303030304e-06, "loss": 0.0014, "num_tokens": 4091580.0, "reward": 0.6754500865936279, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942752122879028, "reward_meter_std": 0.0037329280748963356, "reward_repeat_penalty_mean": 0.8152778148651123, "reward_repeat_penalty_std": 0.05002203211188316, "reward_std": 0.04033275321125984, "reward_total_composite_mean": 0.6754500865936279, "reward_total_composite_std": 0.04033276066184044, "reward_total_mean": 0.6754500865936279, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942752122879028, "rewards/meter/std": 0.0037329280748963356, "rewards/repeat_penalty/mean": 0.8152778148651123, "rewards/repeat_penalty/std": 0.05002203211188316, "rewards/total_composite/mean": 0.6754500865936279, "rewards/total_composite/std": 0.04033276066184044, "sampling/importance_sampling_ratio/max": 1.8784008026123047, "sampling/importance_sampling_ratio/mean": 1.0022673606872559, "sampling/importance_sampling_ratio/min": 0.05193512886762619, "sampling/sampling_logp_difference/max": 2.9577598571777344, "sampling/sampling_logp_difference/mean": 0.019572407007217407, "step": 1815 }, { "clip_ratio/high_max": 0.03601037664338946, "clip_ratio/high_mean": 0.03601037664338946, "clip_ratio/low_mean": 0.022342659067362547, "clip_ratio/low_min": 0.022342659067362547, "clip_ratio/region_mean": 0.05835303571075201, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 51.375, "completions/mean_terminated_length": 51.375, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.709318995475769, "epoch": 0.07294051492147648, "frac_reward_zero_std": 0.0, "grad_norm": 7.065499782562256, "learning_rate": 4.5e-06, "loss": 0.3243, "num_tokens": 4093095.0, "reward": 0.6230374574661255, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.5175492167472839, "reward_meter_mean": 0.9962137937545776, "reward_meter_std": 0.0032140789553523064, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.5159306526184082, "reward_total_composite_mean": 0.6230374574661255, "reward_total_composite_std": 0.5159306526184082, "reward_total_mean": 0.6230374574661255, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.5175492167472839, "rewards/meter/mean": 0.9962137937545776, "rewards/meter/std": 0.0032140789553523064, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6230374574661255, "rewards/total_composite/std": 0.5159306526184082, "sampling/importance_sampling_ratio/max": 1.9115855693817139, "sampling/importance_sampling_ratio/mean": 1.0141249895095825, "sampling/importance_sampling_ratio/min": 0.3503313660621643, "sampling/sampling_logp_difference/max": 1.0488758087158203, "sampling/sampling_logp_difference/mean": 0.08005251735448837, "step": 1816 }, { "clip_ratio/high_max": 0.050623598508536816, "clip_ratio/high_mean": 0.050623598508536816, "clip_ratio/low_mean": 0.011928289197385311, "clip_ratio/low_min": 0.011928289197385311, "clip_ratio/region_mean": 0.06255188770592213, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 72.625, "completions/mean_terminated_length": 72.625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.8766130320727825, "epoch": 0.07298068040326143, "frac_reward_zero_std": 0.0, "grad_norm": 9.501838684082031, "learning_rate": 4.496969696969697e-06, "loss": 0.0293, "num_tokens": 4094892.0, "reward": 0.8915443420410156, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8915443420410156, "reward_meter_std": 0.21483440697193146, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.21483437716960907, "reward_total_composite_mean": 0.8915443420410156, "reward_total_composite_std": 0.21483440697193146, "reward_total_mean": 0.8915443420410156, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8915443420410156, "rewards/meter/std": 0.21483440697193146, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8915443420410156, "rewards/total_composite/std": 0.21483440697193146, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.027990460395813, "sampling/importance_sampling_ratio/min": 0.11607716232538223, "sampling/sampling_logp_difference/max": 2.1535000801086426, "sampling/sampling_logp_difference/mean": 0.08924838155508041, "step": 1817 }, { "clip_ratio/high_max": 0.0036526747280731797, "clip_ratio/high_mean": 0.0036526747280731797, "clip_ratio/low_mean": 0.006177184404805303, "clip_ratio/low_min": 0.006177184404805303, "clip_ratio/region_mean": 0.009829859132878482, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 102.0, "completions/mean_terminated_length": 102.0, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.04315004916861653, "epoch": 0.07302084588504638, "frac_reward_zero_std": 0.0, "grad_norm": 0.42083650827407837, "learning_rate": 4.493939393939395e-06, "loss": -0.0031, "num_tokens": 4097204.0, "reward": 0.9974691271781921, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974691271781921, "reward_meter_std": 6.317263614619151e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.316715735010803e-05, "reward_total_composite_mean": 0.9974691271781921, "reward_total_composite_std": 6.317263614619151e-05, "reward_total_mean": 0.9974691271781921, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974691271781921, "rewards/meter/std": 6.317263614619151e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974691271781921, "rewards/total_composite/std": 6.317263614619151e-05, "sampling/importance_sampling_ratio/max": 1.6696717739105225, "sampling/importance_sampling_ratio/mean": 1.001711130142212, "sampling/importance_sampling_ratio/min": 0.3871941566467285, "sampling/sampling_logp_difference/max": 0.9488290548324585, "sampling/sampling_logp_difference/mean": 0.009578624740242958, "step": 1818 }, { "clip_ratio/high_max": 0.020909853279590607, "clip_ratio/high_mean": 0.020909853279590607, "clip_ratio/low_mean": 0.006778868613764644, "clip_ratio/low_min": 0.006778868613764644, "clip_ratio/region_mean": 0.02768872189335525, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 93.75, "completions/mean_terminated_length": 93.75, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.23089196532964706, "epoch": 0.07306101136683134, "frac_reward_zero_std": 0.0, "grad_norm": 2.0940115451812744, "learning_rate": 4.490909090909091e-06, "loss": -0.0058, "num_tokens": 4099258.0, "reward": 0.8056372404098511, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8540617823600769, "reward_meter_std": 0.2617916762828827, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.25278210639953613, "reward_total_composite_mean": 0.8056372404098511, "reward_total_composite_std": 0.25278210639953613, "reward_total_mean": 0.8056372404098511, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8540617823600769, "rewards/meter/std": 0.2617916762828827, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8056372404098511, "rewards/total_composite/std": 0.25278210639953613, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0129790306091309, "sampling/importance_sampling_ratio/min": 0.3954644799232483, "sampling/sampling_logp_difference/max": 0.9276943206787109, "sampling/sampling_logp_difference/mean": 0.029167180880904198, "step": 1819 }, { "clip_ratio/high_max": 0.008658499689772725, "clip_ratio/high_mean": 0.008658499689772725, "clip_ratio/low_mean": 0.006392460549250245, "clip_ratio/low_min": 0.006392460549250245, "clip_ratio/region_mean": 0.01505096023902297, "completions/clipped_ratio": 0.0, "completions/max_length": 59.0, "completions/max_terminated_length": 59.0, "completions/mean_length": 57.875, "completions/mean_terminated_length": 57.875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.10346860811114311, "epoch": 0.0731011768486163, "frac_reward_zero_std": 0.0, "grad_norm": 2.0801198482513428, "learning_rate": 4.487878787878788e-06, "loss": 0.0047, "num_tokens": 4101001.0, "reward": 0.9942807555198669, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942807555198669, "reward_meter_std": 0.0004856908926740289, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00048569359933026135, "reward_total_composite_mean": 0.9942807555198669, "reward_total_composite_std": 0.0004856908926740289, "reward_total_mean": 0.9942807555198669, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942807555198669, "rewards/meter/std": 0.0004856908926740289, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942807555198669, "rewards/total_composite/std": 0.0004856908926740289, "sampling/importance_sampling_ratio/max": 1.5113404989242554, "sampling/importance_sampling_ratio/mean": 1.0053722858428955, "sampling/importance_sampling_ratio/min": 0.2805930972099304, "sampling/sampling_logp_difference/max": 1.2708497047424316, "sampling/sampling_logp_difference/mean": 0.017049286514520645, "step": 1820 }, { "clip_ratio/high_max": 0.040383500047028065, "clip_ratio/high_mean": 0.040383500047028065, "clip_ratio/low_mean": 0.016557018272578716, "clip_ratio/low_min": 0.016557018272578716, "clip_ratio/region_mean": 0.05694051831960678, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 74.5, "completions/mean_terminated_length": 74.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.6388323791325092, "epoch": 0.07314134233040126, "frac_reward_zero_std": 0.0, "grad_norm": 5.139247417449951, "learning_rate": 4.4848484848484855e-06, "loss": 0.0157, "num_tokens": 4102797.0, "reward": 0.9978154897689819, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978154897689819, "reward_meter_std": 0.0017045722343027592, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0017045725835487247, "reward_total_composite_mean": 0.9978154897689819, "reward_total_composite_std": 0.0017045722343027592, "reward_total_mean": 0.9978154897689819, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978154897689819, "rewards/meter/std": 0.0017045722343027592, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978154897689819, "rewards/total_composite/std": 0.0017045722343027592, "sampling/importance_sampling_ratio/max": 1.806069016456604, "sampling/importance_sampling_ratio/mean": 1.015410304069519, "sampling/importance_sampling_ratio/min": 0.346730500459671, "sampling/sampling_logp_difference/max": 1.0592074394226074, "sampling/sampling_logp_difference/mean": 0.055102791637182236, "step": 1821 }, { "clip_ratio/high_max": 0.018180417479015887, "clip_ratio/high_mean": 0.018180417479015887, "clip_ratio/low_mean": 0.020897203590720892, "clip_ratio/low_min": 0.020897203590720892, "clip_ratio/region_mean": 0.03907762106973678, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 73.75, "completions/mean_terminated_length": 73.75, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.5206275209784508, "epoch": 0.07318150781218621, "frac_reward_zero_std": 0.0, "grad_norm": 5.289615154266357, "learning_rate": 4.481818181818182e-06, "loss": -0.0115, "num_tokens": 4104667.0, "reward": 0.9970654249191284, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970654249191284, "reward_meter_std": 0.0017285288777202368, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0017285366775467992, "reward_total_composite_mean": 0.9970654249191284, "reward_total_composite_std": 0.0017285288777202368, "reward_total_mean": 0.9970654249191284, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970654249191284, "rewards/meter/std": 0.0017285288777202368, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970654249191284, "rewards/total_composite/std": 0.0017285288777202368, "sampling/importance_sampling_ratio/max": 1.728150725364685, "sampling/importance_sampling_ratio/mean": 1.0059843063354492, "sampling/importance_sampling_ratio/min": 0.30036142468452454, "sampling/sampling_logp_difference/max": 1.2027688026428223, "sampling/sampling_logp_difference/mean": 0.05023278295993805, "step": 1822 }, { "clip_ratio/high_max": 0.007692307699471712, "clip_ratio/high_mean": 0.007692307699471712, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007692307699471712, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.5, "completions/mean_terminated_length": 64.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.033252993831411004, "epoch": 0.07322167329397117, "frac_reward_zero_std": 0.0, "grad_norm": 2.6084089279174805, "learning_rate": 4.478787878787879e-06, "loss": -0.0029, "num_tokens": 4106511.0, "reward": 0.9984657764434814, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984657764434814, "reward_meter_std": 0.0002891587500926107, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002891614567488432, "reward_total_composite_mean": 0.9984657764434814, "reward_total_composite_std": 0.0002891587500926107, "reward_total_mean": 0.9984657764434814, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984657764434814, "rewards/meter/std": 0.0002891587500926107, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984657764434814, "rewards/total_composite/std": 0.0002891587500926107, "sampling/importance_sampling_ratio/max": 1.241689920425415, "sampling/importance_sampling_ratio/mean": 0.9986276626586914, "sampling/importance_sampling_ratio/min": 0.6341525316238403, "sampling/sampling_logp_difference/max": 0.45546579360961914, "sampling/sampling_logp_difference/mean": 0.007291331887245178, "step": 1823 }, { "clip_ratio/high_max": 0.044933649245649576, "clip_ratio/high_mean": 0.044933649245649576, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.04850507783703506, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 37.75, "completions/mean_terminated_length": 37.75, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.39845724031329155, "epoch": 0.07326183877575612, "frac_reward_zero_std": 0.0, "grad_norm": 11.332904815673828, "learning_rate": 4.4757575757575765e-06, "loss": -0.0206, "num_tokens": 4108093.0, "reward": 0.9941079616546631, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9941079616546631, "reward_meter_std": 0.012935124337673187, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.012935123406350613, "reward_total_composite_mean": 0.9941079616546631, "reward_total_composite_std": 0.012935124337673187, "reward_total_mean": 0.9941079616546631, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9941079616546631, "rewards/meter/std": 0.012935124337673187, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9941079616546631, "rewards/total_composite/std": 0.012935124337673187, "sampling/importance_sampling_ratio/max": 1.575133204460144, "sampling/importance_sampling_ratio/mean": 1.0034513473510742, "sampling/importance_sampling_ratio/min": 0.25851887464523315, "sampling/sampling_logp_difference/max": 1.3527865409851074, "sampling/sampling_logp_difference/mean": 0.0540194995701313, "step": 1824 }, { "clip_ratio/high_max": 0.0026315790601074696, "clip_ratio/high_mean": 0.0026315790601074696, "clip_ratio/low_mean": 0.02371319057419896, "clip_ratio/low_min": 0.02371319057419896, "clip_ratio/region_mean": 0.02634476963430643, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 94.625, "completions/mean_terminated_length": 94.625, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.3725603432394564, "epoch": 0.07330200425754108, "frac_reward_zero_std": 0.0, "grad_norm": 4.047023296356201, "learning_rate": 4.472727272727273e-06, "loss": 0.0106, "num_tokens": 4110202.0, "reward": 0.554187536239624, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.587390661239624, "reward_meter_std": 0.44364675879478455, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.42821574211120605, "reward_total_composite_mean": 0.554187536239624, "reward_total_composite_std": 0.42821574211120605, "reward_total_mean": 0.554187536239624, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.587390661239624, "rewards/meter/std": 0.44364675879478455, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.554187536239624, "rewards/total_composite/std": 0.42821574211120605, "sampling/importance_sampling_ratio/max": 1.9479297399520874, "sampling/importance_sampling_ratio/mean": 1.0154355764389038, "sampling/importance_sampling_ratio/min": 0.37140175700187683, "sampling/sampling_logp_difference/max": 0.9904708862304688, "sampling/sampling_logp_difference/mean": 0.03696272522211075, "step": 1825 }, { "clip_ratio/high_max": 0.004032257944345474, "clip_ratio/high_mean": 0.004032257944345474, "clip_ratio/low_mean": 0.0066495382925495505, "clip_ratio/low_min": 0.0066495382925495505, "clip_ratio/region_mean": 0.010681796236895025, "completions/clipped_ratio": 0.0, "completions/max_length": 95.0, "completions/max_terminated_length": 95.0, "completions/mean_length": 92.75, "completions/mean_terminated_length": 92.75, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.20322295650839806, "epoch": 0.07334216973932603, "frac_reward_zero_std": 0.0, "grad_norm": 3.477172613143921, "learning_rate": 4.46969696969697e-06, "loss": 0.0078, "num_tokens": 4112208.0, "reward": 0.8535131216049194, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9020949602127075, "reward_meter_std": 0.12145007401704788, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.12295857816934586, "reward_total_composite_mean": 0.8535131216049194, "reward_total_composite_std": 0.12295859307050705, "reward_total_mean": 0.8535131216049194, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9020949602127075, "rewards/meter/std": 0.12145007401704788, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8535131216049194, "rewards/total_composite/std": 0.12295859307050705, "sampling/importance_sampling_ratio/max": 1.6056605577468872, "sampling/importance_sampling_ratio/mean": 1.0086493492126465, "sampling/importance_sampling_ratio/min": 0.3523665964603424, "sampling/sampling_logp_difference/max": 1.0430831909179688, "sampling/sampling_logp_difference/mean": 0.02582640014588833, "step": 1826 }, { "clip_ratio/high_max": 0.03203791263513267, "clip_ratio/high_mean": 0.03203791263513267, "clip_ratio/low_mean": 0.015318386955186725, "clip_ratio/low_min": 0.015318386955186725, "clip_ratio/region_mean": 0.047356299590319395, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 100.125, "completions/mean_terminated_length": 100.125, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.37415438517928123, "epoch": 0.07338233522111098, "frac_reward_zero_std": 0.0, "grad_norm": 3.58325457572937, "learning_rate": 4.4666666666666665e-06, "loss": -0.0104, "num_tokens": 4114241.0, "reward": 0.7074779868125916, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7311835289001465, "reward_meter_std": 0.286437064409256, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.27346280217170715, "reward_total_composite_mean": 0.7074779868125916, "reward_total_composite_std": 0.27346280217170715, "reward_total_mean": 0.7074779868125916, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7311835289001465, "rewards/meter/std": 0.286437064409256, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7074779868125916, "rewards/total_composite/std": 0.27346280217170715, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.018310546875, "sampling/importance_sampling_ratio/min": 0.4192293882369995, "sampling/sampling_logp_difference/max": 0.8693370819091797, "sampling/sampling_logp_difference/mean": 0.04709630832076073, "step": 1827 }, { "clip_ratio/high_max": 0.007329145446419716, "clip_ratio/high_mean": 0.007329145446419716, "clip_ratio/low_mean": 0.0012254902394488454, "clip_ratio/low_min": 0.0012254902394488454, "clip_ratio/region_mean": 0.008554635685868561, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 101.75, "completions/mean_terminated_length": 101.75, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.06690388498827815, "epoch": 0.07342250070289594, "frac_reward_zero_std": 0.0, "grad_norm": 1.8939130306243896, "learning_rate": 4.463636363636364e-06, "loss": 0.0007, "num_tokens": 4116559.0, "reward": 0.9971822500228882, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971822500228882, "reward_meter_std": 0.0007533096941187978, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007533166790381074, "reward_total_composite_mean": 0.9971822500228882, "reward_total_composite_std": 0.0007533096941187978, "reward_total_mean": 0.9971822500228882, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971822500228882, "rewards/meter/std": 0.0007533096941187978, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971822500228882, "rewards/total_composite/std": 0.0007533096941187978, "sampling/importance_sampling_ratio/max": 1.46595299243927, "sampling/importance_sampling_ratio/mean": 1.002759337425232, "sampling/importance_sampling_ratio/min": 0.5041558742523193, "sampling/sampling_logp_difference/max": 0.6848697662353516, "sampling/sampling_logp_difference/mean": 0.011065986938774586, "step": 1828 }, { "clip_ratio/high_max": 0.0033174321288242936, "clip_ratio/high_mean": 0.0033174321288242936, "clip_ratio/low_mean": 0.002019958512391895, "clip_ratio/low_min": 0.002019958512391895, "clip_ratio/region_mean": 0.005337390641216189, "completions/clipped_ratio": 0.0, "completions/max_length": 189.0, "completions/max_terminated_length": 189.0, "completions/mean_length": 185.75, "completions/mean_terminated_length": 185.75, "completions/min_length": 183.0, "completions/min_terminated_length": 183.0, "entropy": 0.10350964730605483, "epoch": 0.07346266618468089, "frac_reward_zero_std": 0.0, "grad_norm": 6.232636451721191, "learning_rate": 4.460606060606061e-06, "loss": -0.0037, "num_tokens": 4119597.0, "reward": 0.6353453397750854, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9841963052749634, "reward_meter_std": 0.033278584480285645, "reward_repeat_penalty_mean": 0.7750000357627869, "reward_repeat_penalty_std": 0.0707106813788414, "reward_std": 0.05924128741025925, "reward_total_composite_mean": 0.6353453397750854, "reward_total_composite_std": 0.05924127250909805, "reward_total_mean": 0.6353453397750854, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9841963052749634, "rewards/meter/std": 0.033278584480285645, "rewards/repeat_penalty/mean": 0.7750000357627869, "rewards/repeat_penalty/std": 0.0707106813788414, "rewards/total_composite/mean": 0.6353453397750854, "rewards/total_composite/std": 0.05924127250909805, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0010361671447754, "sampling/importance_sampling_ratio/min": 0.222689688205719, "sampling/sampling_logp_difference/max": 1.5019760131835938, "sampling/sampling_logp_difference/mean": 0.0172612052410841, "step": 1829 }, { "clip_ratio/high_max": 0.0012626262614503503, "clip_ratio/high_mean": 0.0012626262614503503, "clip_ratio/low_mean": 0.0012626262614503503, "clip_ratio/low_min": 0.0012626262614503503, "clip_ratio/region_mean": 0.0025252525229007006, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 99.0, "completions/mean_terminated_length": 99.0, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.020989181706681848, "epoch": 0.07350283166646585, "frac_reward_zero_std": 0.0, "grad_norm": 0.13791632652282715, "learning_rate": 4.4575757575757575e-06, "loss": 0.0001, "num_tokens": 4121701.0, "reward": 0.9990502595901489, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990502595901489, "reward_meter_std": 1.3007632333028596e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.300759322475642e-05, "reward_total_composite_mean": 0.9990502595901489, "reward_total_composite_std": 1.3007632333028596e-05, "reward_total_mean": 0.9990502595901489, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990502595901489, "rewards/meter/std": 1.3007632333028596e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990502595901489, "rewards/total_composite/std": 1.3007632333028596e-05, "sampling/importance_sampling_ratio/max": 1.6697462797164917, "sampling/importance_sampling_ratio/mean": 1.0018277168273926, "sampling/importance_sampling_ratio/min": 0.6270684599876404, "sampling/sampling_logp_difference/max": 0.512671709060669, "sampling/sampling_logp_difference/mean": 0.00393805792555213, "step": 1830 }, { "clip_ratio/high_max": 0.0055555556900799274, "clip_ratio/high_mean": 0.0055555556900799274, "clip_ratio/low_mean": 0.010989011265337467, "clip_ratio/low_min": 0.010989011265337467, "clip_ratio/region_mean": 0.016544566955417395, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 89.5, "completions/mean_terminated_length": 89.5, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.14992150850594044, "epoch": 0.0735429971482508, "frac_reward_zero_std": 0.0, "grad_norm": 3.588621139526367, "learning_rate": 4.454545454545455e-06, "loss": 0.0079, "num_tokens": 4123721.0, "reward": 0.9949823021888733, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949823021888733, "reward_meter_std": 0.0008774827001616359, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008774733869358897, "reward_total_composite_mean": 0.9949823021888733, "reward_total_composite_std": 0.0008774827001616359, "reward_total_mean": 0.9949823021888733, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949823021888733, "rewards/meter/std": 0.0008774827001616359, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949823021888733, "rewards/total_composite/std": 0.0008774827001616359, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007697343826294, "sampling/importance_sampling_ratio/min": 0.3661029040813446, "sampling/sampling_logp_difference/max": 1.0048408508300781, "sampling/sampling_logp_difference/mean": 0.02186514623463154, "step": 1831 }, { "clip_ratio/high_max": 0.0055555556900799274, "clip_ratio/high_mean": 0.0055555556900799274, "clip_ratio/low_mean": 0.008428030530922115, "clip_ratio/low_min": 0.008428030530922115, "clip_ratio/region_mean": 0.013983586221002042, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 89.75, "completions/mean_terminated_length": 89.75, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.09856001799926162, "epoch": 0.07358316263003575, "frac_reward_zero_std": 0.0, "grad_norm": 1.7238988876342773, "learning_rate": 4.451515151515152e-06, "loss": 0.0014, "num_tokens": 4125807.0, "reward": 0.9952454566955566, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952454566955566, "reward_meter_std": 0.00010376512364018708, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010377082071499899, "reward_total_composite_mean": 0.9952454566955566, "reward_total_composite_std": 0.00010376512364018708, "reward_total_mean": 0.9952454566955566, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952454566955566, "rewards/meter/std": 0.00010376512364018708, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952454566955566, "rewards/total_composite/std": 0.00010376512364018708, "sampling/importance_sampling_ratio/max": 1.3277589082717896, "sampling/importance_sampling_ratio/mean": 1.001192331314087, "sampling/importance_sampling_ratio/min": 0.2499622255563736, "sampling/sampling_logp_difference/max": 1.3864455223083496, "sampling/sampling_logp_difference/mean": 0.014209388755261898, "step": 1832 }, { "clip_ratio/high_max": 0.016408111667260528, "clip_ratio/high_mean": 0.016408111667260528, "clip_ratio/low_mean": 0.018417607876472175, "clip_ratio/low_min": 0.018417607876472175, "clip_ratio/region_mean": 0.0348257195437327, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 75.5, "completions/mean_terminated_length": 75.5, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.36707402765750885, "epoch": 0.07362332811182071, "frac_reward_zero_std": 0.0, "grad_norm": 3.797224283218384, "learning_rate": 4.448484848484848e-06, "loss": 0.0092, "num_tokens": 4127747.0, "reward": 0.9976780414581299, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976780414581299, "reward_meter_std": 0.0009906893828883767, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009906899649649858, "reward_total_composite_mean": 0.9976780414581299, "reward_total_composite_std": 0.0009906893828883767, "reward_total_mean": 0.9976780414581299, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976780414581299, "rewards/meter/std": 0.0009906893828883767, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976780414581299, "rewards/total_composite/std": 0.0009906893828883767, "sampling/importance_sampling_ratio/max": 1.881462812423706, "sampling/importance_sampling_ratio/mean": 1.0070687532424927, "sampling/importance_sampling_ratio/min": 0.4079129993915558, "sampling/sampling_logp_difference/max": 0.8967013359069824, "sampling/sampling_logp_difference/mean": 0.03965409845113754, "step": 1833 }, { "clip_ratio/high_max": 0.03248523222282529, "clip_ratio/high_mean": 0.03248523222282529, "clip_ratio/low_mean": 0.015092328016180545, "clip_ratio/low_min": 0.015092328016180545, "clip_ratio/region_mean": 0.047577560239005834, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 133.625, "completions/mean_terminated_length": 133.625, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.501116044819355, "epoch": 0.07366349359360566, "frac_reward_zero_std": 0.0, "grad_norm": 3.744053840637207, "learning_rate": 4.445454545454546e-06, "loss": -0.0051, "num_tokens": 4130264.0, "reward": 0.9247440099716187, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959684014320374, "reward_meter_std": 0.0031451100949198008, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.07493869215250015, "reward_total_composite_mean": 0.9247440099716187, "reward_total_composite_std": 0.07493867725133896, "reward_total_mean": 0.9247440099716187, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959684014320374, "rewards/meter/std": 0.0031451100949198008, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9247440099716187, "rewards/total_composite/std": 0.07493867725133896, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0151046514511108, "sampling/importance_sampling_ratio/min": 0.1339227855205536, "sampling/sampling_logp_difference/max": 2.0104918479919434, "sampling/sampling_logp_difference/mean": 0.06190461292862892, "step": 1834 }, { "clip_ratio/high_max": 0.023937532445415854, "clip_ratio/high_mean": 0.023937532445415854, "clip_ratio/low_mean": 0.008216594811528921, "clip_ratio/low_min": 0.008216594811528921, "clip_ratio/region_mean": 0.032154127256944776, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.375, "completions/mean_terminated_length": 62.375, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.26288264244794846, "epoch": 0.07370365907539062, "frac_reward_zero_std": 0.0, "grad_norm": 5.053841590881348, "learning_rate": 4.442424242424243e-06, "loss": -0.0192, "num_tokens": 4132027.0, "reward": 0.8166337013244629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8166337013244629, "reward_meter_std": 0.3090912401676178, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3090912103652954, "reward_total_composite_mean": 0.8166337013244629, "reward_total_composite_std": 0.3090912401676178, "reward_total_mean": 0.8166337013244629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8166337013244629, "rewards/meter/std": 0.3090912401676178, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8166337013244629, "rewards/total_composite/std": 0.3090912401676178, "sampling/importance_sampling_ratio/max": 1.7029978036880493, "sampling/importance_sampling_ratio/mean": 0.9984228610992432, "sampling/importance_sampling_ratio/min": 0.2625430226325989, "sampling/sampling_logp_difference/max": 1.3373403549194336, "sampling/sampling_logp_difference/mean": 0.038962990045547485, "step": 1835 }, { "clip_ratio/high_max": 0.04100045142695308, "clip_ratio/high_mean": 0.04100045142695308, "clip_ratio/low_mean": 0.00966143561527133, "clip_ratio/low_min": 0.00966143561527133, "clip_ratio/region_mean": 0.05066188704222441, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 116.5, "completions/mean_terminated_length": 116.5, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.6946443617343903, "epoch": 0.07374382455717557, "frac_reward_zero_std": 0.0, "grad_norm": 5.317627906799316, "learning_rate": 4.43939393939394e-06, "loss": 0.0132, "num_tokens": 4134279.0, "reward": 0.9559872150421143, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9806186556816101, "reward_meter_std": 0.017093582078814507, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06989037245512009, "reward_total_composite_mean": 0.9559872150421143, "reward_total_composite_std": 0.06989036500453949, "reward_total_mean": 0.9559872150421143, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9806186556816101, "rewards/meter/std": 0.017093582078814507, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9559872150421143, "rewards/total_composite/std": 0.06989036500453949, "sampling/importance_sampling_ratio/max": 1.9605425596237183, "sampling/importance_sampling_ratio/mean": 1.0162169933319092, "sampling/importance_sampling_ratio/min": 0.30436596274375916, "sampling/sampling_logp_difference/max": 1.1895244121551514, "sampling/sampling_logp_difference/mean": 0.07094138115644455, "step": 1836 }, { "clip_ratio/high_max": 0.0029411765281111, "clip_ratio/high_mean": 0.0029411765281111, "clip_ratio/low_mean": 0.000735294132027775, "clip_ratio/low_min": 0.000735294132027775, "clip_ratio/region_mean": 0.0036764706601388752, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 170.0, "completions/mean_terminated_length": 170.0, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.04040496004745364, "epoch": 0.07378399003896052, "frac_reward_zero_std": 0.0, "grad_norm": 1.2245804071426392, "learning_rate": 4.436363636363637e-06, "loss": 0.0003, "num_tokens": 4137119.0, "reward": 0.7593173980712891, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958264231681824, "reward_meter_std": 6.799297989346087e-05, "reward_repeat_penalty_mean": 0.762499988079071, "reward_repeat_penalty_std": 0.05175492912530899, "reward_std": 0.051534902304410934, "reward_total_composite_mean": 0.7593173980712891, "reward_total_composite_std": 0.05153491348028183, "reward_total_mean": 0.7593173980712891, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958264231681824, "rewards/meter/std": 6.799297989346087e-05, "rewards/repeat_penalty/mean": 0.762499988079071, "rewards/repeat_penalty/std": 0.05175492912530899, "rewards/total_composite/mean": 0.7593173980712891, "rewards/total_composite/std": 0.05153491348028183, "sampling/importance_sampling_ratio/max": 1.4358134269714355, "sampling/importance_sampling_ratio/mean": 1.0017375946044922, "sampling/importance_sampling_ratio/min": 0.5273821353912354, "sampling/sampling_logp_difference/max": 0.6398299336433411, "sampling/sampling_logp_difference/mean": 0.005329389125108719, "step": 1837 }, { "clip_ratio/high_max": 0.022940170136280358, "clip_ratio/high_mean": 0.022940170136280358, "clip_ratio/low_mean": 0.022611827589571476, "clip_ratio/low_min": 0.022611827589571476, "clip_ratio/region_mean": 0.045551997725851834, "completions/clipped_ratio": 0.0, "completions/max_length": 240.0, "completions/max_terminated_length": 240.0, "completions/mean_length": 228.5, "completions/mean_terminated_length": 228.5, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.668959241360426, "epoch": 0.07382415552074548, "frac_reward_zero_std": 0.0, "grad_norm": 3.743595838546753, "learning_rate": 4.433333333333334e-06, "loss": -0.019, "num_tokens": 4140579.0, "reward": 0.7390586137771606, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.7707350254058838, "reward_meter_std": 0.3024085462093353, "reward_repeat_penalty_mean": 0.9791666865348816, "reward_repeat_penalty_std": 0.038575831800699234, "reward_std": 0.29844897985458374, "reward_total_composite_mean": 0.7390586137771606, "reward_total_composite_std": 0.29844897985458374, "reward_total_mean": 0.7390586137771606, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.7707350254058838, "rewards/meter/std": 0.3024085462093353, "rewards/repeat_penalty/mean": 0.9791666865348816, "rewards/repeat_penalty/std": 0.038575831800699234, "rewards/total_composite/mean": 0.7390586137771606, "rewards/total_composite/std": 0.29844897985458374, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.015356183052063, "sampling/importance_sampling_ratio/min": 0.3005274534225464, "sampling/sampling_logp_difference/max": 1.2022161483764648, "sampling/sampling_logp_difference/mean": 0.06499035656452179, "step": 1838 }, { "clip_ratio/high_max": 0.02181874285452068, "clip_ratio/high_mean": 0.02181874285452068, "clip_ratio/low_mean": 0.01651052851229906, "clip_ratio/low_min": 0.01651052851229906, "clip_ratio/region_mean": 0.03832927136681974, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 78.25, "completions/mean_terminated_length": 78.25, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.7259627729654312, "epoch": 0.07386432100253043, "frac_reward_zero_std": 0.0, "grad_norm": 5.02839994430542, "learning_rate": 4.430303030303031e-06, "loss": -0.03, "num_tokens": 4142461.0, "reward": 0.9820351600646973, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9820351600646973, "reward_meter_std": 0.018020369112491608, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.018020374700427055, "reward_total_composite_mean": 0.9820351600646973, "reward_total_composite_std": 0.018020369112491608, "reward_total_mean": 0.9820351600646973, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9820351600646973, "rewards/meter/std": 0.018020369112491608, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9820351600646973, "rewards/total_composite/std": 0.018020369112491608, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0230660438537598, "sampling/importance_sampling_ratio/min": 0.24417172372341156, "sampling/sampling_logp_difference/max": 1.4098834991455078, "sampling/sampling_logp_difference/mean": 0.07023806869983673, "step": 1839 }, { "clip_ratio/high_max": 0.027999775717034936, "clip_ratio/high_mean": 0.027999775717034936, "clip_ratio/low_mean": 0.022959183901548386, "clip_ratio/low_min": 0.022959183901548386, "clip_ratio/region_mean": 0.05095895961858332, "completions/clipped_ratio": 0.0, "completions/max_length": 53.0, "completions/max_terminated_length": 53.0, "completions/mean_length": 49.625, "completions/mean_terminated_length": 49.625, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.24762486666440964, "epoch": 0.07390448648431538, "frac_reward_zero_std": 0.0, "grad_norm": 6.407735347747803, "learning_rate": 4.4272727272727275e-06, "loss": 0.0184, "num_tokens": 4144162.0, "reward": 0.7431851029396057, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7431851029396057, "reward_meter_std": 0.294132262468338, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2941322326660156, "reward_total_composite_mean": 0.7431851029396057, "reward_total_composite_std": 0.294132262468338, "reward_total_mean": 0.7431851029396057, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7431851029396057, "rewards/meter/std": 0.294132262468338, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7431851029396057, "rewards/total_composite/std": 0.294132262468338, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0074049234390259, "sampling/importance_sampling_ratio/min": 0.2499392181634903, "sampling/sampling_logp_difference/max": 1.3865375518798828, "sampling/sampling_logp_difference/mean": 0.05073566362261772, "step": 1840 }, { "clip_ratio/high_max": 0.00737299001775682, "clip_ratio/high_mean": 0.00737299001775682, "clip_ratio/low_mean": 0.008309659664519131, "clip_ratio/low_min": 0.008309659664519131, "clip_ratio/region_mean": 0.01568264968227595, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 119.875, "completions/mean_terminated_length": 119.875, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.08843067381531, "epoch": 0.07394465196610034, "frac_reward_zero_std": 0.0, "grad_norm": 3.6605427265167236, "learning_rate": 4.424242424242425e-06, "loss": 0.012, "num_tokens": 4146697.0, "reward": 0.9058225154876709, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946545362472534, "reward_meter_std": 0.0017242503818124533, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07321520894765854, "reward_total_composite_mean": 0.9058225154876709, "reward_total_composite_std": 0.07321521639823914, "reward_total_mean": 0.9058225154876709, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946545362472534, "rewards/meter/std": 0.0017242503818124533, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9058225154876709, "rewards/total_composite/std": 0.07321521639823914, "sampling/importance_sampling_ratio/max": 1.6830854415893555, "sampling/importance_sampling_ratio/mean": 0.9984482526779175, "sampling/importance_sampling_ratio/min": 0.34670352935791016, "sampling/sampling_logp_difference/max": 1.0592851638793945, "sampling/sampling_logp_difference/mean": 0.016406936571002007, "step": 1841 }, { "clip_ratio/high_max": 0.00787545822095126, "clip_ratio/high_mean": 0.00787545822095126, "clip_ratio/low_mean": 0.012987380847334862, "clip_ratio/low_min": 0.012987380847334862, "clip_ratio/region_mean": 0.02086283906828612, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.2709883898496628, "epoch": 0.07398481744788529, "frac_reward_zero_std": 0.0, "grad_norm": 4.1317853927612305, "learning_rate": 4.421212121212122e-06, "loss": 0.0062, "num_tokens": 4148545.0, "reward": 0.8449044227600098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8449044227600098, "reward_meter_std": 0.19496634602546692, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.19496633112430573, "reward_total_composite_mean": 0.8449044227600098, "reward_total_composite_std": 0.19496634602546692, "reward_total_mean": 0.8449044227600098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8449044227600098, "rewards/meter/std": 0.19496634602546692, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8449044227600098, "rewards/total_composite/std": 0.19496634602546692, "sampling/importance_sampling_ratio/max": 1.8676364421844482, "sampling/importance_sampling_ratio/mean": 1.0063512325286865, "sampling/importance_sampling_ratio/min": 0.2587898373603821, "sampling/sampling_logp_difference/max": 1.3517389297485352, "sampling/sampling_logp_difference/mean": 0.03727368265390396, "step": 1842 }, { "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/low_mean": 0.005357142887078226, "clip_ratio/low_min": 0.005357142887078226, "clip_ratio/region_mean": 0.010791925597004592, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.375, "completions/mean_terminated_length": 69.375, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.09468106366693974, "epoch": 0.07402498292967025, "frac_reward_zero_std": 0.0, "grad_norm": 2.7024903297424316, "learning_rate": 4.418181818181818e-06, "loss": 0.0033, "num_tokens": 4150316.0, "reward": 0.9967325925827026, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9967325925827026, "reward_meter_std": 0.0019061544444411993, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0019061638740822673, "reward_total_composite_mean": 0.9967325925827026, "reward_total_composite_std": 0.0019061544444411993, "reward_total_mean": 0.9967325925827026, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9967325925827026, "rewards/meter/std": 0.0019061544444411993, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967325925827026, "rewards/total_composite/std": 0.0019061544444411993, "sampling/importance_sampling_ratio/max": 1.3403892517089844, "sampling/importance_sampling_ratio/mean": 1.0016083717346191, "sampling/importance_sampling_ratio/min": 0.23633739352226257, "sampling/sampling_logp_difference/max": 1.4424948692321777, "sampling/sampling_logp_difference/mean": 0.015712035819888115, "step": 1843 }, { "clip_ratio/high_max": 0.01599274785257876, "clip_ratio/high_mean": 0.01599274785257876, "clip_ratio/low_mean": 0.00394825276453048, "clip_ratio/low_min": 0.00394825276453048, "clip_ratio/region_mean": 0.01994100061710924, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 94.0, "completions/mean_terminated_length": 94.0, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.2582856100052595, "epoch": 0.0740651484114552, "frac_reward_zero_std": 0.0, "grad_norm": 4.094827175140381, "learning_rate": 4.415151515151516e-06, "loss": 0.0055, "num_tokens": 4152604.0, "reward": 0.9527285695075989, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9527285695075989, "reward_meter_std": 0.020587077364325523, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02058705873787403, "reward_total_composite_mean": 0.9527285695075989, "reward_total_composite_std": 0.020587077364325523, "reward_total_mean": 0.9527285695075989, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9527285695075989, "rewards/meter/std": 0.020587077364325523, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9527285695075989, "rewards/total_composite/std": 0.020587077364325523, "sampling/importance_sampling_ratio/max": 1.6599549055099487, "sampling/importance_sampling_ratio/mean": 1.006330132484436, "sampling/importance_sampling_ratio/min": 0.06561143696308136, "sampling/sampling_logp_difference/max": 2.7240052223205566, "sampling/sampling_logp_difference/mean": 0.0326327309012413, "step": 1844 }, { "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005434782709926367, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.07721387594938278, "epoch": 0.07410531389324015, "frac_reward_zero_std": 0.0, "grad_norm": 6.061025619506836, "learning_rate": 4.412121212121213e-06, "loss": -0.0044, "num_tokens": 4154452.0, "reward": 0.9951037168502808, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9951037168502808, "reward_meter_std": 0.0059710158966481686, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005971014499664307, "reward_total_composite_mean": 0.9951037168502808, "reward_total_composite_std": 0.0059710158966481686, "reward_total_mean": 0.9951037168502808, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9951037168502808, "rewards/meter/std": 0.0059710158966481686, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951037168502808, "rewards/total_composite/std": 0.0059710158966481686, "sampling/importance_sampling_ratio/max": 1.1276072263717651, "sampling/importance_sampling_ratio/mean": 1.0022023916244507, "sampling/importance_sampling_ratio/min": 0.5560474395751953, "sampling/sampling_logp_difference/max": 0.5869016647338867, "sampling/sampling_logp_difference/mean": 0.009792874567210674, "step": 1845 }, { "clip_ratio/high_max": 0.015674303169362247, "clip_ratio/high_mean": 0.015674303169362247, "clip_ratio/low_mean": 0.00973088713362813, "clip_ratio/low_min": 0.00973088713362813, "clip_ratio/region_mean": 0.025405190302990377, "completions/clipped_ratio": 0.0, "completions/max_length": 156.0, "completions/max_terminated_length": 156.0, "completions/mean_length": 152.25, "completions/mean_terminated_length": 152.25, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "entropy": 0.28932092152535915, "epoch": 0.07414547937502511, "frac_reward_zero_std": 0.0, "grad_norm": 2.195159912109375, "learning_rate": 4.409090909090909e-06, "loss": 0.0137, "num_tokens": 4157062.0, "reward": 0.7611056566238403, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8805509805679321, "reward_meter_std": 0.11502306908369064, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.05143444612622261, "reward_std": 0.12567628920078278, "reward_total_composite_mean": 0.7611056566238403, "reward_total_composite_std": 0.12567628920078278, "reward_total_mean": 0.7611056566238403, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8805509805679321, "rewards/meter/std": 0.11502306908369064, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.7611056566238403, "rewards/total_composite/std": 0.12567628920078278, "sampling/importance_sampling_ratio/max": 1.9269274473190308, "sampling/importance_sampling_ratio/mean": 1.0067634582519531, "sampling/importance_sampling_ratio/min": 0.27968674898147583, "sampling/sampling_logp_difference/max": 1.2740850448608398, "sampling/sampling_logp_difference/mean": 0.02993069775402546, "step": 1846 }, { "clip_ratio/high_max": 0.0018115942366421223, "clip_ratio/high_mean": 0.0018115942366421223, "clip_ratio/low_mean": 0.0027372262557037175, "clip_ratio/low_min": 0.0027372262557037175, "clip_ratio/region_mean": 0.00454882049234584, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 137.25, "completions/mean_terminated_length": 137.25, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.058579874224960804, "epoch": 0.07418564485681006, "frac_reward_zero_std": 0.0, "grad_norm": 1.8122365474700928, "learning_rate": 4.4060606060606066e-06, "loss": 0.002, "num_tokens": 4159504.0, "reward": 0.8709865212440491, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956631660461426, "reward_meter_std": 0.004949868656694889, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.04542805254459381, "reward_total_composite_mean": 0.8709865212440491, "reward_total_composite_std": 0.04542805626988411, "reward_total_mean": 0.8709865212440491, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956631660461426, "rewards/meter/std": 0.004949868656694889, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8709865212440491, "rewards/total_composite/std": 0.04542805626988411, "sampling/importance_sampling_ratio/max": 1.290838360786438, "sampling/importance_sampling_ratio/mean": 1.0003859996795654, "sampling/importance_sampling_ratio/min": 0.36321109533309937, "sampling/sampling_logp_difference/max": 1.0127711296081543, "sampling/sampling_logp_difference/mean": 0.00922767911106348, "step": 1847 }, { "clip_ratio/high_max": 0.03866560058668256, "clip_ratio/high_mean": 0.03866560058668256, "clip_ratio/low_mean": 0.01064768130891025, "clip_ratio/low_min": 0.01064768130891025, "clip_ratio/region_mean": 0.04931328189559281, "completions/clipped_ratio": 0.0, "completions/max_length": 226.0, "completions/max_terminated_length": 226.0, "completions/mean_length": 193.5, "completions/mean_terminated_length": 193.5, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.6156773902475834, "epoch": 0.07422581033859502, "frac_reward_zero_std": 0.0, "grad_norm": 3.0020229816436768, "learning_rate": 4.403030303030304e-06, "loss": -0.0158, "num_tokens": 4162692.0, "reward": 0.9245690107345581, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9245690107345581, "reward_meter_std": 0.12092939764261246, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.12092941999435425, "reward_total_composite_mean": 0.9245690107345581, "reward_total_composite_std": 0.12092939764261246, "reward_total_mean": 0.9245690107345581, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9245690107345581, "rewards/meter/std": 0.12092939764261246, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9245690107345581, "rewards/total_composite/std": 0.12092939764261246, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01200270652771, "sampling/importance_sampling_ratio/min": 0.24066513776779175, "sampling/sampling_logp_difference/max": 1.4243488311767578, "sampling/sampling_logp_difference/mean": 0.06919539719820023, "step": 1848 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0012135922443121672, "clip_ratio/low_min": 0.0012135922443121672, "clip_ratio/region_mean": 0.0012135922443121672, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 103.0, "completions/mean_terminated_length": 103.0, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.04185265023261309, "epoch": 0.07426597582037997, "frac_reward_zero_std": 0.0, "grad_norm": 0.7777861952781677, "learning_rate": 4.4e-06, "loss": 0.0019, "num_tokens": 4165004.0, "reward": 0.9962564706802368, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962564706802368, "reward_meter_std": 0.003385263029485941, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0033852593041956425, "reward_total_composite_mean": 0.9962564706802368, "reward_total_composite_std": 0.003385263029485941, "reward_total_mean": 0.9962564706802368, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962564706802368, "rewards/meter/std": 0.003385263029485941, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9962564706802368, "rewards/total_composite/std": 0.003385263029485941, "sampling/importance_sampling_ratio/max": 1.859822154045105, "sampling/importance_sampling_ratio/mean": 1.0040092468261719, "sampling/importance_sampling_ratio/min": 0.5530079007148743, "sampling/sampling_logp_difference/max": 0.6204808950424194, "sampling/sampling_logp_difference/mean": 0.006807548459619284, "step": 1849 }, { "clip_ratio/high_max": 0.011363636702299118, "clip_ratio/high_mean": 0.011363636702299118, "clip_ratio/low_mean": 0.005741003900766373, "clip_ratio/low_min": 0.005741003900766373, "clip_ratio/region_mean": 0.01710464060306549, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.75, "completions/mean_terminated_length": 65.75, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.06614176463335752, "epoch": 0.07430614130216492, "frac_reward_zero_std": 0.0, "grad_norm": 5.336696624755859, "learning_rate": 4.3969696969696975e-06, "loss": -0.0062, "num_tokens": 4166714.0, "reward": 0.9984281063079834, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984281063079834, "reward_meter_std": 0.0004244510782882571, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004244422307237983, "reward_total_composite_mean": 0.9984281063079834, "reward_total_composite_std": 0.0004244510782882571, "reward_total_mean": 0.9984281063079834, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984281063079834, "rewards/meter/std": 0.0004244510782882571, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984281063079834, "rewards/total_composite/std": 0.0004244510782882571, "sampling/importance_sampling_ratio/max": 1.7678344249725342, "sampling/importance_sampling_ratio/mean": 0.9998464584350586, "sampling/importance_sampling_ratio/min": 0.2822016477584839, "sampling/sampling_logp_difference/max": 1.2651333808898926, "sampling/sampling_logp_difference/mean": 0.02056110091507435, "step": 1850 }, { "epoch": 0.07430614130216492, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 327.2307692307692, "eval_completions/max_terminated_length": 327.2307692307692, "eval_completions/mean_length": 193.1153846153846, "eval_completions/mean_terminated_length": 193.1153846153846, "eval_completions/min_length": 63.76923076923077, "eval_completions/min_terminated_length": 63.76923076923077, "eval_entropy": 0.2572394087910652, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 4166714.0, "eval_reward": 0.510624709037634, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.8968340616959792, "eval_reward_count_adherence_std": 0.13571923226118088, "eval_reward_meter_mean": 0.6479656077348269, "eval_reward_meter_std": 0.4532018624819242, "eval_reward_repeat_penalty_mean": 0.8583002182153555, "eval_reward_repeat_penalty_std": 0.13356439138834292, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.510624709037634, "eval_reward_total_composite_std": 0.38263802230358124, "eval_reward_total_mean": 0.510624709037634, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.8968340616959792, "eval_rewards/count_adherence/std": 0.13571923226118088, "eval_rewards/meter/mean": 0.6479656077348269, "eval_rewards/meter/std": 0.4532018624819242, "eval_rewards/repeat_penalty/mean": 0.8583002182153555, "eval_rewards/repeat_penalty/std": 0.13356439138834292, "eval_rewards/total_composite/mean": 0.510624709037634, "eval_rewards/total_composite/std": 0.38263802230358124, "eval_runtime": 62.689, "eval_samples_per_second": 1.659, "eval_sampling/importance_sampling_ratio/max": 1.4861979392858653, "eval_sampling/importance_sampling_ratio/mean": 1.0072369300402129, "eval_sampling/importance_sampling_ratio/min": 0.3671792642428325, "eval_sampling/sampling_logp_difference/max": 1.0324598183998694, "eval_sampling/sampling_logp_difference/mean": 0.023962400352152493, "eval_steps_per_second": 0.207, "step": 1850 }, { "clip_ratio/high_max": 0.00905440840870142, "clip_ratio/high_mean": 0.00905440840870142, "clip_ratio/low_mean": 0.044483832316473126, "clip_ratio/low_min": 0.044483832316473126, "clip_ratio/region_mean": 0.053538240725174546, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 110.875, "completions/mean_terminated_length": 110.875, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.5153284594416618, "epoch": 0.07434630678394988, "frac_reward_zero_std": 0.0, "grad_norm": 4.0140767097473145, "learning_rate": 4.393939393939394e-06, "loss": 0.0113, "num_tokens": 4168913.0, "reward": 0.37161165475845337, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.37161165475845337, "reward_meter_std": 0.27739158272743225, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.27739158272743225, "reward_total_composite_mean": 0.37161165475845337, "reward_total_composite_std": 0.27739158272743225, "reward_total_mean": 0.37161165475845337, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.37161165475845337, "rewards/meter/std": 0.27739158272743225, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.37161165475845337, "rewards/total_composite/std": 0.27739158272743225, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0157358646392822, "sampling/importance_sampling_ratio/min": 0.2537955343723297, "sampling/sampling_logp_difference/max": 1.3712263107299805, "sampling/sampling_logp_difference/mean": 0.06999441236257553, "step": 1851 }, { "clip_ratio/high_max": 0.04731742758303881, "clip_ratio/high_mean": 0.04731742758303881, "clip_ratio/low_mean": 0.01895719300955534, "clip_ratio/low_min": 0.01895719300955534, "clip_ratio/region_mean": 0.06627462059259415, "completions/clipped_ratio": 0.0, "completions/max_length": 273.0, "completions/max_terminated_length": 273.0, "completions/mean_length": 247.75, "completions/mean_terminated_length": 247.75, "completions/min_length": 222.0, "completions/min_terminated_length": 222.0, "entropy": 0.7441882863640785, "epoch": 0.07438647226573483, "frac_reward_zero_std": 0.0, "grad_norm": 3.773585557937622, "learning_rate": 4.390909090909091e-06, "loss": 0.0023, "num_tokens": 4172359.0, "reward": 0.6160822510719299, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9464285969734192, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.6540890336036682, "reward_meter_std": 0.43809670209884644, "reward_repeat_penalty_mean": 0.9910714626312256, "reward_repeat_penalty_std": 0.025253823027014732, "reward_std": 0.42173710465431213, "reward_total_composite_mean": 0.6160822510719299, "reward_total_composite_std": 0.42173710465431213, "reward_total_mean": 0.6160822510719299, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9464285969734192, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.6540890336036682, "rewards/meter/std": 0.43809670209884644, "rewards/repeat_penalty/mean": 0.9910714626312256, "rewards/repeat_penalty/std": 0.025253823027014732, "rewards/total_composite/mean": 0.6160822510719299, "rewards/total_composite/std": 0.42173710465431213, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.012731671333313, "sampling/importance_sampling_ratio/min": 0.002913407515734434, "sampling/sampling_logp_difference/max": 5.8384318351745605, "sampling/sampling_logp_difference/mean": 0.09465256333351135, "step": 1852 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.0035714285913854837, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 35.375, "completions/mean_terminated_length": 35.375, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.18969010561704636, "epoch": 0.07442663774751979, "frac_reward_zero_std": 0.0, "grad_norm": 11.522212982177734, "learning_rate": 4.387878787878788e-06, "loss": 0.006, "num_tokens": 4174010.0, "reward": 0.9776397943496704, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9776397943496704, "reward_meter_std": 0.01006159745156765, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010061599314212799, "reward_total_composite_mean": 0.9776397943496704, "reward_total_composite_std": 0.01006159745156765, "reward_total_mean": 0.9776397943496704, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9776397943496704, "rewards/meter/std": 0.01006159745156765, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9776397943496704, "rewards/total_composite/std": 0.01006159745156765, "sampling/importance_sampling_ratio/max": 1.7054202556610107, "sampling/importance_sampling_ratio/mean": 1.0117498636245728, "sampling/importance_sampling_ratio/min": 0.6540494561195374, "sampling/sampling_logp_difference/max": 0.5338115692138672, "sampling/sampling_logp_difference/mean": 0.01866281032562256, "step": 1853 }, { "clip_ratio/high_max": 0.03236477100290358, "clip_ratio/high_mean": 0.03236477100290358, "clip_ratio/low_mean": 0.009980237111449242, "clip_ratio/low_min": 0.009980237111449242, "clip_ratio/region_mean": 0.04234500811435282, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 114.625, "completions/mean_terminated_length": 114.625, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.5414317175745964, "epoch": 0.07446680322930474, "frac_reward_zero_std": 0.0, "grad_norm": 4.39829158782959, "learning_rate": 4.384848484848485e-06, "loss": 0.0082, "num_tokens": 4176279.0, "reward": 0.9875234365463257, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9875234365463257, "reward_meter_std": 0.01168668083846569, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01168669294565916, "reward_total_composite_mean": 0.9875234365463257, "reward_total_composite_std": 0.01168668083846569, "reward_total_mean": 0.9875234365463257, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9875234365463257, "rewards/meter/std": 0.01168668083846569, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9875234365463257, "rewards/total_composite/std": 0.01168668083846569, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0194075107574463, "sampling/importance_sampling_ratio/min": 0.27460548281669617, "sampling/sampling_logp_difference/max": 1.2924197912216187, "sampling/sampling_logp_difference/mean": 0.054695580154657364, "step": 1854 }, { "clip_ratio/high_max": 0.008928571827709675, "clip_ratio/high_mean": 0.008928571827709675, "clip_ratio/low_mean": 0.0022321429569274187, "clip_ratio/low_min": 0.0022321429569274187, "clip_ratio/region_mean": 0.011160714784637094, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 56.0, "completions/mean_terminated_length": 56.0, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.0864438284188509, "epoch": 0.0745069687110897, "frac_reward_zero_std": 0.0, "grad_norm": 2.625075101852417, "learning_rate": 4.381818181818182e-06, "loss": 0.0017, "num_tokens": 4178031.0, "reward": 0.9880209565162659, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9880209565162659, "reward_meter_std": 0.007883170619606972, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0078831622377038, "reward_total_composite_mean": 0.9880209565162659, "reward_total_composite_std": 0.007883170619606972, "reward_total_mean": 0.9880209565162659, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9880209565162659, "rewards/meter/std": 0.007883170619606972, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9880209565162659, "rewards/total_composite/std": 0.007883170619606972, "sampling/importance_sampling_ratio/max": 1.396715521812439, "sampling/importance_sampling_ratio/mean": 1.0048121213912964, "sampling/importance_sampling_ratio/min": 0.46448272466659546, "sampling/sampling_logp_difference/max": 0.7668309211730957, "sampling/sampling_logp_difference/mean": 0.010663346387445927, "step": 1855 }, { "clip_ratio/high_max": 0.023573311744257808, "clip_ratio/high_mean": 0.023573311744257808, "clip_ratio/low_mean": 0.007754115387797356, "clip_ratio/low_min": 0.007754115387797356, "clip_ratio/region_mean": 0.03132742713205516, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 111.875, "completions/mean_terminated_length": 111.875, "completions/min_length": 109.0, "completions/min_terminated_length": 109.0, "entropy": 0.35822908766567707, "epoch": 0.07454713419287465, "frac_reward_zero_std": 0.0, "grad_norm": 3.003889799118042, "learning_rate": 4.378787878787879e-06, "loss": 0.0108, "num_tokens": 4180182.0, "reward": 0.8977147340774536, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974768757820129, "reward_meter_std": 0.0010170461609959602, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.10649988800287247, "reward_total_composite_mean": 0.8977147340774536, "reward_total_composite_std": 0.10649988800287247, "reward_total_mean": 0.8977147340774536, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974768757820129, "rewards/meter/std": 0.0010170461609959602, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8977147340774536, "rewards/total_composite/std": 0.10649988800287247, "sampling/importance_sampling_ratio/max": 1.5614416599273682, "sampling/importance_sampling_ratio/mean": 1.004133701324463, "sampling/importance_sampling_ratio/min": 0.2102976143360138, "sampling/sampling_logp_difference/max": 1.5592315196990967, "sampling/sampling_logp_difference/mean": 0.04684070870280266, "step": 1856 }, { "clip_ratio/high_max": 0.0021551724057644606, "clip_ratio/high_mean": 0.0021551724057644606, "clip_ratio/low_mean": 0.017778822453692555, "clip_ratio/low_min": 0.017778822453692555, "clip_ratio/region_mean": 0.019933994859457016, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 56.75, "completions/mean_terminated_length": 56.75, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.09877756051719189, "epoch": 0.0745872996746596, "frac_reward_zero_std": 0.0, "grad_norm": 1.938165307044983, "learning_rate": 4.375757575757576e-06, "loss": -0.0105, "num_tokens": 4181844.0, "reward": 0.9916501641273499, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9916501641273499, "reward_meter_std": 0.001931402483023703, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0019314087694510818, "reward_total_composite_mean": 0.9916501641273499, "reward_total_composite_std": 0.001931402483023703, "reward_total_mean": 0.9916501641273499, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9916501641273499, "rewards/meter/std": 0.001931402483023703, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9916501641273499, "rewards/total_composite/std": 0.001931402483023703, "sampling/importance_sampling_ratio/max": 1.4867905378341675, "sampling/importance_sampling_ratio/mean": 1.0090504884719849, "sampling/importance_sampling_ratio/min": 0.3470149040222168, "sampling/sampling_logp_difference/max": 1.0583875179290771, "sampling/sampling_logp_difference/mean": 0.013949423097074032, "step": 1857 }, { "clip_ratio/high_max": 0.007095551351085305, "clip_ratio/high_mean": 0.007095551351085305, "clip_ratio/low_mean": 0.0028572361916303635, "clip_ratio/low_min": 0.0028572361916303635, "clip_ratio/region_mean": 0.009952787542715669, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 88.25, "completions/mean_terminated_length": 88.25, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.09635706804692745, "epoch": 0.07462746515644456, "frac_reward_zero_std": 0.0, "grad_norm": 2.222642421722412, "learning_rate": 4.372727272727273e-06, "loss": -0.0057, "num_tokens": 4183958.0, "reward": 0.9945516586303711, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9945516586303711, "reward_meter_std": 0.0012705601984634995, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0012705654371529818, "reward_total_composite_mean": 0.9945516586303711, "reward_total_composite_std": 0.0012705601984634995, "reward_total_mean": 0.9945516586303711, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9945516586303711, "rewards/meter/std": 0.0012705601984634995, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9945516586303711, "rewards/total_composite/std": 0.0012705601984634995, "sampling/importance_sampling_ratio/max": 1.6056901216506958, "sampling/importance_sampling_ratio/mean": 1.0038167238235474, "sampling/importance_sampling_ratio/min": 0.4266884922981262, "sampling/sampling_logp_difference/max": 0.851701021194458, "sampling/sampling_logp_difference/mean": 0.012971967458724976, "step": 1858 }, { "clip_ratio/high_max": 0.0049019609577953815, "clip_ratio/high_mean": 0.0049019609577953815, "clip_ratio/low_mean": 0.007352941436693072, "clip_ratio/low_min": 0.007352941436693072, "clip_ratio/region_mean": 0.012254902394488454, "completions/clipped_ratio": 0.0, "completions/max_length": 51.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 51.0, "completions/mean_terminated_length": 51.0, "completions/min_length": 51.0, "completions/min_terminated_length": 51.0, "entropy": 0.049468324054032564, "epoch": 0.07466763063822951, "frac_reward_zero_std": 0.0, "grad_norm": 4.79885721206665, "learning_rate": 4.36969696969697e-06, "loss": -0.0001, "num_tokens": 4185702.0, "reward": 0.9259762763977051, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9259762763977051, "reward_meter_std": 0.0015891619259491563, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001589166815392673, "reward_total_composite_mean": 0.9259762763977051, "reward_total_composite_std": 0.0015891619259491563, "reward_total_mean": 0.9259762763977051, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9259762763977051, "rewards/meter/std": 0.0015891619259491563, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9259762763977051, "rewards/total_composite/std": 0.0015891619259491563, "sampling/importance_sampling_ratio/max": 1.6099872589111328, "sampling/importance_sampling_ratio/mean": 0.9998874068260193, "sampling/importance_sampling_ratio/min": 0.4403989613056183, "sampling/sampling_logp_difference/max": 0.820074200630188, "sampling/sampling_logp_difference/mean": 0.012182224541902542, "step": 1859 }, { "clip_ratio/high_max": 0.006407182663679123, "clip_ratio/high_mean": 0.006407182663679123, "clip_ratio/low_mean": 0.008354876074008644, "clip_ratio/low_min": 0.008354876074008644, "clip_ratio/region_mean": 0.014762058737687767, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 135.0, "completions/mean_terminated_length": 135.0, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.100110930390656, "epoch": 0.07470779612001446, "frac_reward_zero_std": 0.0, "grad_norm": 2.1606459617614746, "learning_rate": 4.366666666666667e-06, "loss": -0.0032, "num_tokens": 4188182.0, "reward": 0.8834315538406372, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.989547610282898, "reward_meter_std": 0.007958509027957916, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06423989683389664, "reward_total_composite_mean": 0.8834315538406372, "reward_total_composite_std": 0.06423989683389664, "reward_total_mean": 0.8834315538406372, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.989547610282898, "rewards/meter/std": 0.007958509027957916, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8834315538406372, "rewards/total_composite/std": 0.06423989683389664, "sampling/importance_sampling_ratio/max": 1.8788100481033325, "sampling/importance_sampling_ratio/mean": 1.0032159090042114, "sampling/importance_sampling_ratio/min": 0.30122095346450806, "sampling/sampling_logp_difference/max": 1.199911117553711, "sampling/sampling_logp_difference/mean": 0.01665782928466797, "step": 1860 }, { "clip_ratio/high_max": 0.0124762476189062, "clip_ratio/high_mean": 0.0124762476189062, "clip_ratio/low_mean": 0.009549640817567706, "clip_ratio/low_min": 0.009549640817567706, "clip_ratio/region_mean": 0.022025888436473906, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 101.0, "completions/mean_terminated_length": 101.0, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.16979045886546373, "epoch": 0.07474796160179942, "frac_reward_zero_std": 0.0, "grad_norm": 2.7288973331451416, "learning_rate": 4.363636363636364e-06, "loss": 0.0208, "num_tokens": 4190278.0, "reward": 0.7065380811691284, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7360309958457947, "reward_meter_std": 0.27524590492248535, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.27462708950042725, "reward_total_composite_mean": 0.7065380811691284, "reward_total_composite_std": 0.27462705969810486, "reward_total_mean": 0.7065380811691284, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7360309958457947, "rewards/meter/std": 0.27524590492248535, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.7065380811691284, "rewards/total_composite/std": 0.27462705969810486, "sampling/importance_sampling_ratio/max": 1.5228219032287598, "sampling/importance_sampling_ratio/mean": 1.0014276504516602, "sampling/importance_sampling_ratio/min": 0.07940194010734558, "sampling/sampling_logp_difference/max": 2.5332324504852295, "sampling/sampling_logp_difference/mean": 0.0317775197327137, "step": 1861 }, { "clip_ratio/high_max": 0.005077758920378983, "clip_ratio/high_mean": 0.005077758920378983, "clip_ratio/low_mean": 0.0011111111380159855, "clip_ratio/low_min": 0.0011111111380159855, "clip_ratio/region_mean": 0.0061888700583949685, "completions/clipped_ratio": 0.0, "completions/max_length": 225.0, "completions/max_terminated_length": 225.0, "completions/mean_length": 221.875, "completions/mean_terminated_length": 221.875, "completions/min_length": 221.0, "completions/min_terminated_length": 221.0, "entropy": 0.03992933081462979, "epoch": 0.07478812708358437, "frac_reward_zero_std": 0.0, "grad_norm": 1.0823556184768677, "learning_rate": 4.36060606060606e-06, "loss": 0.0047, "num_tokens": 4193661.0, "reward": 0.6643049716949463, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9964575171470642, "reward_meter_std": 0.0009934601839631796, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000662296952214092, "reward_total_composite_mean": 0.6643049716949463, "reward_total_composite_std": 0.0006622962537221611, "reward_total_mean": 0.6643049716949463, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9964575171470642, "rewards/meter/std": 0.0009934601839631796, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6643049716949463, "rewards/total_composite/std": 0.0006622962537221611, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0002436637878418, "sampling/importance_sampling_ratio/min": 7.986734999576584e-05, "sampling/sampling_logp_difference/max": 9.43514347076416, "sampling/sampling_logp_difference/mean": 0.01341662835329771, "step": 1862 }, { "clip_ratio/high_max": 0.05100303632207215, "clip_ratio/high_mean": 0.05100303632207215, "clip_ratio/low_mean": 0.006666666828095913, "clip_ratio/low_min": 0.006666666828095913, "clip_ratio/region_mean": 0.05766970315016806, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 73.25, "completions/mean_terminated_length": 73.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.6060812920331955, "epoch": 0.07482829256536933, "frac_reward_zero_std": 0.0, "grad_norm": 5.034853458404541, "learning_rate": 4.3575757575757576e-06, "loss": 0.0087, "num_tokens": 4195615.0, "reward": 0.9600696563720703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9600696563720703, "reward_meter_std": 0.09926661849021912, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09926661849021912, "reward_total_composite_mean": 0.9600696563720703, "reward_total_composite_std": 0.09926661849021912, "reward_total_mean": 0.9600696563720703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9600696563720703, "rewards/meter/std": 0.09926661849021912, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9600696563720703, "rewards/total_composite/std": 0.09926661849021912, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0097596645355225, "sampling/importance_sampling_ratio/min": 0.08340954035520554, "sampling/sampling_logp_difference/max": 2.483992576599121, "sampling/sampling_logp_difference/mean": 0.07004112005233765, "step": 1863 }, { "clip_ratio/high_max": 0.0013888889225199819, "clip_ratio/high_mean": 0.0013888889225199819, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0013888889225199819, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 89.75, "completions/mean_terminated_length": 89.75, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.05834774626418948, "epoch": 0.07486845804715428, "frac_reward_zero_std": 0.0, "grad_norm": 1.9747538566589355, "learning_rate": 4.354545454545455e-06, "loss": -0.0061, "num_tokens": 4197677.0, "reward": 0.9951763153076172, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9951763153076172, "reward_meter_std": 0.0005524771986529231, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005524645675905049, "reward_total_composite_mean": 0.9951763153076172, "reward_total_composite_std": 0.0005524771986529231, "reward_total_mean": 0.9951763153076172, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9951763153076172, "rewards/meter/std": 0.0005524771986529231, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951763153076172, "rewards/total_composite/std": 0.0005524771986529231, "sampling/importance_sampling_ratio/max": 1.4608083963394165, "sampling/importance_sampling_ratio/mean": 1.0047131776809692, "sampling/importance_sampling_ratio/min": 0.34022247791290283, "sampling/sampling_logp_difference/max": 1.078155517578125, "sampling/sampling_logp_difference/mean": 0.008235161192715168, "step": 1864 }, { "clip_ratio/high_max": 0.0036231884732842445, "clip_ratio/high_mean": 0.0036231884732842445, "clip_ratio/low_mean": 0.010716472752392292, "clip_ratio/low_min": 0.010716472752392292, "clip_ratio/region_mean": 0.014339661225676537, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.375, "completions/mean_terminated_length": 69.375, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.12075466383248568, "epoch": 0.07490862352893923, "frac_reward_zero_std": 0.0, "grad_norm": 3.9621639251708984, "learning_rate": 4.351515151515152e-06, "loss": 0.0111, "num_tokens": 4199512.0, "reward": 0.9972849488258362, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972849488258362, "reward_meter_std": 0.0004286824550945312, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000428676517913118, "reward_total_composite_mean": 0.9972849488258362, "reward_total_composite_std": 0.0004286824550945312, "reward_total_mean": 0.9972849488258362, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972849488258362, "rewards/meter/std": 0.0004286824550945312, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972849488258362, "rewards/total_composite/std": 0.0004286824550945312, "sampling/importance_sampling_ratio/max": 1.5455111265182495, "sampling/importance_sampling_ratio/mean": 1.003419280052185, "sampling/importance_sampling_ratio/min": 0.4426264464855194, "sampling/sampling_logp_difference/max": 0.8150291442871094, "sampling/sampling_logp_difference/mean": 0.015396162867546082, "step": 1865 }, { "clip_ratio/high_max": 0.0036407767329365015, "clip_ratio/high_mean": 0.0036407767329365015, "clip_ratio/low_mean": 0.0036407767329365015, "clip_ratio/low_min": 0.0036407767329365015, "clip_ratio/region_mean": 0.007281553465873003, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 103.375, "completions/mean_terminated_length": 103.375, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.03981838142499328, "epoch": 0.07494878901072419, "frac_reward_zero_std": 0.0, "grad_norm": 1.6598762273788452, "learning_rate": 4.348484848484849e-06, "loss": 0.0048, "num_tokens": 4201587.0, "reward": 0.9974133968353271, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974133968353271, "reward_meter_std": 0.00014544985606335104, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00014544471923727542, "reward_total_composite_mean": 0.9974133968353271, "reward_total_composite_std": 0.00014544985606335104, "reward_total_mean": 0.9974133968353271, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974133968353271, "rewards/meter/std": 0.00014544985606335104, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974133968353271, "rewards/total_composite/std": 0.00014544985606335104, "sampling/importance_sampling_ratio/max": 1.4277647733688354, "sampling/importance_sampling_ratio/mean": 0.9997120499610901, "sampling/importance_sampling_ratio/min": 0.3747345805168152, "sampling/sampling_logp_difference/max": 0.9815373420715332, "sampling/sampling_logp_difference/mean": 0.009199890308082104, "step": 1866 }, { "clip_ratio/high_max": 0.01905520213767886, "clip_ratio/high_mean": 0.01905520213767886, "clip_ratio/low_mean": 0.01117079914547503, "clip_ratio/low_min": 0.01117079914547503, "clip_ratio/region_mean": 0.03022600128315389, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 356.625, "completions/mean_terminated_length": 356.625, "completions/min_length": 331.0, "completions/min_terminated_length": 331.0, "entropy": 0.6004703845828772, "epoch": 0.07498895449250914, "frac_reward_zero_std": 0.0, "grad_norm": 2.981818675994873, "learning_rate": 4.345454545454546e-06, "loss": 0.0447, "num_tokens": 4206088.0, "reward": 0.46806132793426514, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6333333253860474, "reward_count_adherence_std": 0.035634830594062805, "reward_meter_mean": 0.904925525188446, "reward_meter_std": 0.24405907094478607, "reward_repeat_penalty_mean": 0.8377193212509155, "reward_repeat_penalty_std": 0.11382605135440826, "reward_std": 0.1208736002445221, "reward_total_composite_mean": 0.46806132793426514, "reward_total_composite_std": 0.1208736002445221, "reward_total_mean": 0.46806132793426514, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6333333253860474, "rewards/count_adherence/std": 0.035634830594062805, "rewards/meter/mean": 0.904925525188446, "rewards/meter/std": 0.24405907094478607, "rewards/repeat_penalty/mean": 0.8377193212509155, "rewards/repeat_penalty/std": 0.11382605135440826, "rewards/total_composite/mean": 0.46806132793426514, "rewards/total_composite/std": 0.1208736002445221, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0076172351837158, "sampling/importance_sampling_ratio/min": 4.1646697468422644e-07, "sampling/sampling_logp_difference/max": 14.691458702087402, "sampling/sampling_logp_difference/mean": 0.05986325070261955, "step": 1867 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0028093435103073716, "clip_ratio/low_min": 0.0028093435103073716, "clip_ratio/region_mean": 0.0028093435103073716, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 89.875, "completions/mean_terminated_length": 89.875, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.06600452493876219, "epoch": 0.0750291199742941, "frac_reward_zero_std": 0.0, "grad_norm": 0.8906328678131104, "learning_rate": 4.342424242424243e-06, "loss": -0.0004, "num_tokens": 4208151.0, "reward": 0.9952363967895508, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952363967895508, "reward_meter_std": 0.00016089908604044467, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001608873571967706, "reward_total_composite_mean": 0.9952363967895508, "reward_total_composite_std": 0.00016089908604044467, "reward_total_mean": 0.9952363967895508, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952363967895508, "rewards/meter/std": 0.00016089908604044467, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952363967895508, "rewards/total_composite/std": 0.00016089908604044467, "sampling/importance_sampling_ratio/max": 1.6893128156661987, "sampling/importance_sampling_ratio/mean": 1.002822756767273, "sampling/importance_sampling_ratio/min": 0.4610784351825714, "sampling/sampling_logp_difference/max": 0.7741870880126953, "sampling/sampling_logp_difference/mean": 0.009152467362582684, "step": 1868 }, { "clip_ratio/high_max": 0.0012135922443121672, "clip_ratio/high_mean": 0.0012135922443121672, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0012135922443121672, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 102.875, "completions/mean_terminated_length": 102.875, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.03081246791407466, "epoch": 0.07506928545607905, "frac_reward_zero_std": 0.0, "grad_norm": 2.3382408618927, "learning_rate": 4.33939393939394e-06, "loss": 0.0019, "num_tokens": 4210238.0, "reward": 0.9974008798599243, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974008798599243, "reward_meter_std": 0.00030872138449922204, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00030873037758283317, "reward_total_composite_mean": 0.9974008798599243, "reward_total_composite_std": 0.00030872138449922204, "reward_total_mean": 0.9974008798599243, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974008798599243, "rewards/meter/std": 0.00030872138449922204, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974008798599243, "rewards/total_composite/std": 0.00030872138449922204, "sampling/importance_sampling_ratio/max": 1.2280147075653076, "sampling/importance_sampling_ratio/mean": 0.9990761280059814, "sampling/importance_sampling_ratio/min": 0.23188959062099457, "sampling/sampling_logp_difference/max": 1.461493968963623, "sampling/sampling_logp_difference/mean": 0.009639319032430649, "step": 1869 }, { "clip_ratio/high_max": 0.011222697328776121, "clip_ratio/high_mean": 0.011222697328776121, "clip_ratio/low_mean": 0.02176366886124015, "clip_ratio/low_min": 0.02176366886124015, "clip_ratio/region_mean": 0.03298636619001627, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 76.375, "completions/mean_terminated_length": 76.375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.4157518967986107, "epoch": 0.075109450937864, "frac_reward_zero_std": 0.0, "grad_norm": 4.046680927276611, "learning_rate": 4.336363636363637e-06, "loss": -0.0106, "num_tokens": 4212193.0, "reward": 0.9944717288017273, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9944717288017273, "reward_meter_std": 0.003807853441685438, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003807859495282173, "reward_total_composite_mean": 0.9944717288017273, "reward_total_composite_std": 0.003807853441685438, "reward_total_mean": 0.9944717288017273, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9944717288017273, "rewards/meter/std": 0.003807853441685438, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944717288017273, "rewards/total_composite/std": 0.003807853441685438, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.011504054069519, "sampling/importance_sampling_ratio/min": 0.15385551750659943, "sampling/sampling_logp_difference/max": 1.8717412948608398, "sampling/sampling_logp_difference/mean": 0.04645724594593048, "step": 1870 }, { "clip_ratio/high_max": 0.004167824285104871, "clip_ratio/high_mean": 0.004167824285104871, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/region_mean": 0.006286468356847763, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 59.125, "completions/mean_terminated_length": 59.125, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.057940175756812096, "epoch": 0.07514961641964896, "frac_reward_zero_std": 0.0, "grad_norm": 2.565702438354492, "learning_rate": 4.333333333333334e-06, "loss": -0.0038, "num_tokens": 4213938.0, "reward": 0.9952658414840698, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952658414840698, "reward_meter_std": 0.0007007194799371064, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000700704287737608, "reward_total_composite_mean": 0.9952658414840698, "reward_total_composite_std": 0.0007007194799371064, "reward_total_mean": 0.9952658414840698, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952658414840698, "rewards/meter/std": 0.0007007194799371064, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952658414840698, "rewards/total_composite/std": 0.0007007194799371064, "sampling/importance_sampling_ratio/max": 1.3448500633239746, "sampling/importance_sampling_ratio/mean": 0.996799886226654, "sampling/importance_sampling_ratio/min": 0.461254358291626, "sampling/sampling_logp_difference/max": 0.7738056182861328, "sampling/sampling_logp_difference/mean": 0.013889300636947155, "step": 1871 }, { "clip_ratio/high_max": 0.03716663923114538, "clip_ratio/high_mean": 0.03716663923114538, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.03716663923114538, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 233.0, "completions/mean_length": 294.875, "completions/mean_terminated_length": 222.5, "completions/min_length": 212.0, "completions/min_terminated_length": 212.0, "entropy": 0.5619211308658123, "epoch": 0.07518978190143391, "frac_reward_zero_std": 0.0, "grad_norm": 1.4247676134109497, "learning_rate": 4.330303030303031e-06, "loss": -0.3047, "num_tokens": 4216745.0, "reward": 0.6971731185913086, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.8686028718948364, "reward_meter_std": 0.3345178961753845, "reward_repeat_penalty_mean": 0.9460227489471436, "reward_repeat_penalty_std": 0.06318090111017227, "reward_std": 0.433406800031662, "reward_total_composite_mean": 0.6971731185913086, "reward_total_composite_std": 0.433406800031662, "reward_total_mean": 0.6971731185913086, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.8686028718948364, "rewards/meter/std": 0.3345178961753845, "rewards/repeat_penalty/mean": 0.9460227489471436, "rewards/repeat_penalty/std": 0.06318090111017227, "rewards/total_composite/mean": 0.6971731185913086, "rewards/total_composite/std": 0.433406800031662, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.018416404724121, "sampling/importance_sampling_ratio/min": 0.27683258056640625, "sampling/sampling_logp_difference/max": 1.2843422889709473, "sampling/sampling_logp_difference/mean": 0.0717155784368515, "step": 1872 }, { "clip_ratio/high_max": 0.02799722831696272, "clip_ratio/high_mean": 0.02799722831696272, "clip_ratio/low_mean": 0.0032552082557231188, "clip_ratio/low_min": 0.0032552082557231188, "clip_ratio/region_mean": 0.03125243657268584, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 203.0, "completions/mean_length": 235.75, "completions/mean_terminated_length": 196.2857208251953, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.5523100271821022, "epoch": 0.07522994738321886, "frac_reward_zero_std": 0.0, "grad_norm": 1.2419461011886597, "learning_rate": 4.327272727272728e-06, "loss": -0.2461, "num_tokens": 4219511.0, "reward": 0.851571798324585, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8880919814109802, "reward_meter_std": 0.2659795582294464, "reward_repeat_penalty_mean": 0.9611110687255859, "reward_repeat_penalty_std": 0.07582584023475647, "reward_std": 0.26690801978111267, "reward_total_composite_mean": 0.851571798324585, "reward_total_composite_std": 0.26690804958343506, "reward_total_mean": 0.851571798324585, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8880919814109802, "rewards/meter/std": 0.2659795582294464, "rewards/repeat_penalty/mean": 0.9611110687255859, "rewards/repeat_penalty/std": 0.07582584023475647, "rewards/total_composite/mean": 0.851571798324585, "rewards/total_composite/std": 0.26690804958343506, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0172072649002075, "sampling/importance_sampling_ratio/min": 0.2717590928077698, "sampling/sampling_logp_difference/max": 1.3028392791748047, "sampling/sampling_logp_difference/mean": 0.058800045400857925, "step": 1873 }, { "clip_ratio/high_max": 0.0333196020219475, "clip_ratio/high_mean": 0.0333196020219475, "clip_ratio/low_mean": 0.012383450288325548, "clip_ratio/low_min": 0.012383450288325548, "clip_ratio/region_mean": 0.04570305231027305, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 105.75, "completions/mean_terminated_length": 105.75, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.40878177247941494, "epoch": 0.07527011286500382, "frac_reward_zero_std": 0.0, "grad_norm": 6.65327787399292, "learning_rate": 4.324242424242425e-06, "loss": -0.0071, "num_tokens": 4221629.0, "reward": 0.9133010506629944, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9625563621520996, "reward_meter_std": 0.03560182824730873, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.08394213020801544, "reward_total_composite_mean": 0.9133010506629944, "reward_total_composite_std": 0.08394214510917664, "reward_total_mean": 0.9133010506629944, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9625563621520996, "rewards/meter/std": 0.03560182824730873, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9133010506629944, "rewards/total_composite/std": 0.08394214510917664, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9978815317153931, "sampling/importance_sampling_ratio/min": 0.16954979300498962, "sampling/sampling_logp_difference/max": 1.7746086120605469, "sampling/sampling_logp_difference/mean": 0.06265782564878464, "step": 1874 }, { "clip_ratio/high_max": 0.02583647519350052, "clip_ratio/high_mean": 0.02583647519350052, "clip_ratio/low_mean": 0.008985401596873999, "clip_ratio/low_min": 0.008985401596873999, "clip_ratio/region_mean": 0.03482187679037452, "completions/clipped_ratio": 0.0, "completions/max_length": 221.0, "completions/max_terminated_length": 221.0, "completions/mean_length": 202.375, "completions/mean_terminated_length": 202.375, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.6360356472432613, "epoch": 0.07531027834678877, "frac_reward_zero_std": 0.0, "grad_norm": 3.1858253479003906, "learning_rate": 4.321212121212121e-06, "loss": 0.0297, "num_tokens": 4224968.0, "reward": 0.8535768985748291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.888024091720581, "reward_meter_std": 0.2775907516479492, "reward_repeat_penalty_mean": 0.9608585834503174, "reward_repeat_penalty_std": 0.08052574098110199, "reward_std": 0.29564857482910156, "reward_total_composite_mean": 0.8535768985748291, "reward_total_composite_std": 0.29564857482910156, "reward_total_mean": 0.8535768985748291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.888024091720581, "rewards/meter/std": 0.2775907516479492, "rewards/repeat_penalty/mean": 0.9608585834503174, "rewards/repeat_penalty/std": 0.08052574098110199, "rewards/total_composite/mean": 0.8535768985748291, "rewards/total_composite/std": 0.29564857482910156, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0147325992584229, "sampling/importance_sampling_ratio/min": 0.17934706807136536, "sampling/sampling_logp_difference/max": 1.7184324264526367, "sampling/sampling_logp_difference/mean": 0.05819166079163551, "step": 1875 }, { "clip_ratio/high_max": 0.034461831324733794, "clip_ratio/high_mean": 0.034461831324733794, "clip_ratio/low_mean": 0.010135134682059288, "clip_ratio/low_min": 0.010135134682059288, "clip_ratio/region_mean": 0.04459696600679308, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 72.0, "completions/mean_terminated_length": 72.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.309705201536417, "epoch": 0.07535044382857373, "frac_reward_zero_std": 0.0, "grad_norm": 5.3259077072143555, "learning_rate": 4.3181818181818185e-06, "loss": 0.0171, "num_tokens": 4227160.0, "reward": 0.919043779373169, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.919043779373169, "reward_meter_std": 0.19243919849395752, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.19243919849395752, "reward_total_composite_mean": 0.919043779373169, "reward_total_composite_std": 0.19243919849395752, "reward_total_mean": 0.919043779373169, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.919043779373169, "rewards/meter/std": 0.19243919849395752, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.919043779373169, "rewards/total_composite/std": 0.19243919849395752, "sampling/importance_sampling_ratio/max": 1.5501129627227783, "sampling/importance_sampling_ratio/mean": 1.0002164840698242, "sampling/importance_sampling_ratio/min": 0.310654878616333, "sampling/sampling_logp_difference/max": 1.1690726280212402, "sampling/sampling_logp_difference/mean": 0.04635407403111458, "step": 1876 }, { "clip_ratio/high_max": 0.015678069554269314, "clip_ratio/high_mean": 0.015678069554269314, "clip_ratio/low_mean": 0.015004753833636642, "clip_ratio/low_min": 0.015004753833636642, "clip_ratio/region_mean": 0.030682823387905955, "completions/clipped_ratio": 0.0, "completions/max_length": 227.0, "completions/max_terminated_length": 227.0, "completions/mean_length": 204.75, "completions/mean_terminated_length": 204.75, "completions/min_length": 192.0, "completions/min_terminated_length": 192.0, "entropy": 0.5658724009990692, "epoch": 0.07539060931035868, "frac_reward_zero_std": 0.0, "grad_norm": 2.582993745803833, "learning_rate": 4.315151515151516e-06, "loss": -0.0123, "num_tokens": 4230422.0, "reward": 0.9254087209701538, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9647862911224365, "reward_meter_std": 0.05925486981868744, "reward_repeat_penalty_mean": 0.9597222208976746, "reward_repeat_penalty_std": 0.055694278329610825, "reward_std": 0.07272466272115707, "reward_total_composite_mean": 0.9254087209701538, "reward_total_composite_std": 0.07272467762231827, "reward_total_mean": 0.9254087209701538, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9647862911224365, "rewards/meter/std": 0.05925486981868744, "rewards/repeat_penalty/mean": 0.9597222208976746, "rewards/repeat_penalty/std": 0.055694278329610825, "rewards/total_composite/mean": 0.9254087209701538, "rewards/total_composite/std": 0.07272467762231827, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0190201997756958, "sampling/importance_sampling_ratio/min": 0.1992049664258957, "sampling/sampling_logp_difference/max": 1.6134209632873535, "sampling/sampling_logp_difference/mean": 0.05267081782221794, "step": 1877 }, { "clip_ratio/high_max": 0.032999717397615314, "clip_ratio/high_mean": 0.032999717397615314, "clip_ratio/low_mean": 0.011000885395333171, "clip_ratio/low_min": 0.011000885395333171, "clip_ratio/region_mean": 0.044000602792948484, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 332.0, "completions/mean_terminated_length": 306.2857360839844, "completions/min_length": 285.0, "completions/min_terminated_length": 285.0, "entropy": 0.48640119284391403, "epoch": 0.07543077479214363, "frac_reward_zero_std": 0.0, "grad_norm": 1.26083242893219, "learning_rate": 4.312121212121212e-06, "loss": -0.2576, "num_tokens": 4234422.0, "reward": 0.5566210746765137, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.05345224589109421, "reward_meter_mean": 0.8925167322158813, "reward_meter_std": 0.15122567117214203, "reward_repeat_penalty_mean": 0.919905424118042, "reward_repeat_penalty_std": 0.1122746393084526, "reward_std": 0.25293970108032227, "reward_total_composite_mean": 0.5566210746765137, "reward_total_composite_std": 0.25293970108032227, "reward_total_mean": 0.5566210746765137, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.05345224589109421, "rewards/meter/mean": 0.8925167322158813, "rewards/meter/std": 0.15122567117214203, "rewards/repeat_penalty/mean": 0.919905424118042, "rewards/repeat_penalty/std": 0.1122746393084526, "rewards/total_composite/mean": 0.5566210746765137, "rewards/total_composite/std": 0.25293970108032227, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.012736439704895, "sampling/importance_sampling_ratio/min": 3.277708344739949e-07, "sampling/sampling_logp_difference/max": 14.930951118469238, "sampling/sampling_logp_difference/mean": 0.07019904255867004, "step": 1878 }, { "clip_ratio/high_max": 0.0337692714529112, "clip_ratio/high_mean": 0.0337692714529112, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0337692714529112, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 199.75, "completions/mean_terminated_length": 155.1428680419922, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.3837529495358467, "epoch": 0.07547094027392859, "frac_reward_zero_std": 0.0, "grad_norm": 0.8991881012916565, "learning_rate": 4.309090909090909e-06, "loss": -0.2401, "num_tokens": 4236988.0, "reward": 0.866482138633728, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.8689680099487305, "reward_meter_std": 0.3431207239627838, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3501509428024292, "reward_total_composite_mean": 0.866482138633728, "reward_total_composite_std": 0.3501509428024292, "reward_total_mean": 0.866482138633728, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.8689680099487305, "rewards/meter/std": 0.3431207239627838, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.866482138633728, "rewards/total_composite/std": 0.3501509428024292, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0059314966201782, "sampling/importance_sampling_ratio/min": 0.19056470692157745, "sampling/sampling_logp_difference/max": 1.6577634811401367, "sampling/sampling_logp_difference/mean": 0.052908286452293396, "step": 1879 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.013600267469882965, "epoch": 0.07551110575571354, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.306060606060607e-06, "loss": 0.0, "num_tokens": 4238773.0, "reward": 0.997582197189331, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997582197189331, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.997582197189331, "reward_total_composite_std": 0.0, "reward_total_mean": 0.997582197189331, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997582197189331, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997582197189331, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.1089041233062744, "sampling/importance_sampling_ratio/mean": 1.000409722328186, "sampling/importance_sampling_ratio/min": 0.7844833731651306, "sampling/sampling_logp_difference/max": 0.24272990226745605, "sampling/sampling_logp_difference/mean": 0.0021389967296272516, "step": 1880 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.014390837168321013, "epoch": 0.0755512712374985, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.303030303030303e-06, "loss": 0.0, "num_tokens": 4240565.0, "reward": 0.997582197189331, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997582197189331, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.997582197189331, "reward_total_composite_std": 0.0, "reward_total_mean": 0.997582197189331, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997582197189331, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997582197189331, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.1091729402542114, "sampling/importance_sampling_ratio/mean": 1.0012025833129883, "sampling/importance_sampling_ratio/min": 0.9542538523674011, "sampling/sampling_logp_difference/max": 0.10361464321613312, "sampling/sampling_logp_difference/mean": 0.001675679231993854, "step": 1881 }, { "clip_ratio/high_max": 0.0058139534667134285, "clip_ratio/high_mean": 0.0058139534667134285, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/region_mean": 0.011495771817862988, "completions/clipped_ratio": 0.0, "completions/max_length": 45.0, "completions/max_terminated_length": 45.0, "completions/mean_length": 42.625, "completions/mean_terminated_length": 42.625, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.2804165482521057, "epoch": 0.07559143671928345, "frac_reward_zero_std": 0.0, "grad_norm": 2.6807198524475098, "learning_rate": 4.3e-06, "loss": 0.0199, "num_tokens": 4242138.0, "reward": 0.9950450658798218, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9950450658798218, "reward_meter_std": 0.00384511542506516, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0038451338186860085, "reward_total_composite_mean": 0.9950450658798218, "reward_total_composite_std": 0.00384511542506516, "reward_total_mean": 0.9950450658798218, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9950450658798218, "rewards/meter/std": 0.00384511542506516, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950450658798218, "rewards/total_composite/std": 0.00384511542506516, "sampling/importance_sampling_ratio/max": 1.5244181156158447, "sampling/importance_sampling_ratio/mean": 1.0034741163253784, "sampling/importance_sampling_ratio/min": 0.39778339862823486, "sampling/sampling_logp_difference/max": 0.9218476414680481, "sampling/sampling_logp_difference/mean": 0.03229609876871109, "step": 1882 }, { "clip_ratio/high_max": 0.02487414190545678, "clip_ratio/high_mean": 0.02487414190545678, "clip_ratio/low_mean": 0.007823501946404576, "clip_ratio/low_min": 0.007823501946404576, "clip_ratio/region_mean": 0.03269764385186136, "completions/clipped_ratio": 0.0, "completions/max_length": 166.0, "completions/max_terminated_length": 166.0, "completions/mean_length": 160.625, "completions/mean_terminated_length": 160.625, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.38958779722452164, "epoch": 0.0756316022010684, "frac_reward_zero_std": 0.0, "grad_norm": 2.456183433532715, "learning_rate": 4.296969696969698e-06, "loss": -0.0012, "num_tokens": 4244703.0, "reward": 0.9540842771530151, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9891646504402161, "reward_meter_std": 0.004805354867130518, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06925938278436661, "reward_total_composite_mean": 0.9540842771530151, "reward_total_composite_std": 0.06925938278436661, "reward_total_mean": 0.9540842771530151, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9891646504402161, "rewards/meter/std": 0.004805354867130518, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9540842771530151, "rewards/total_composite/std": 0.06925938278436661, "sampling/importance_sampling_ratio/max": 1.8647836446762085, "sampling/importance_sampling_ratio/mean": 1.0097475051879883, "sampling/importance_sampling_ratio/min": 0.37124526500701904, "sampling/sampling_logp_difference/max": 0.9908924102783203, "sampling/sampling_logp_difference/mean": 0.037685707211494446, "step": 1883 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 74.75, "completions/mean_terminated_length": 74.75, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.2754708882421255, "epoch": 0.07567176768285336, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.293939393939394e-06, "loss": 0.0, "num_tokens": 4246477.0, "reward": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963148832321167, "reward_meter_std": 0.0056813484989106655, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "reward_total_mean": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963148832321167, "rewards/meter/std": 0.0056813484989106655, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003008246421814, "sampling/importance_sampling_ratio/min": 0.21740210056304932, "sampling/sampling_logp_difference/max": 1.5260066986083984, "sampling/sampling_logp_difference/mean": 0.04152185842394829, "step": 1884 }, { "clip_ratio/high_max": 0.01472888421267271, "clip_ratio/high_mean": 0.01472888421267271, "clip_ratio/low_mean": 0.010505896178074181, "clip_ratio/low_min": 0.010505896178074181, "clip_ratio/region_mean": 0.02523478039074689, "completions/clipped_ratio": 0.0, "completions/max_length": 260.0, "completions/max_terminated_length": 260.0, "completions/mean_length": 247.375, "completions/mean_terminated_length": 247.375, "completions/min_length": 234.0, "completions/min_terminated_length": 234.0, "entropy": 0.29367757216095924, "epoch": 0.07571193316463831, "frac_reward_zero_std": 0.0, "grad_norm": 2.4556612968444824, "learning_rate": 4.290909090909091e-06, "loss": -0.0235, "num_tokens": 4250056.0, "reward": 0.8244522213935852, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9464285969734192, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.9978925585746765, "reward_meter_std": 0.00047845288645476103, "reward_repeat_penalty_mean": 0.8693909645080566, "reward_repeat_penalty_std": 0.11218275874853134, "reward_std": 0.14857426285743713, "reward_total_composite_mean": 0.8244522213935852, "reward_total_composite_std": 0.14857427775859833, "reward_total_mean": 0.8244522213935852, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9464285969734192, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.9978925585746765, "rewards/meter/std": 0.00047845288645476103, "rewards/repeat_penalty/mean": 0.8693909645080566, "rewards/repeat_penalty/std": 0.11218275874853134, "rewards/total_composite/mean": 0.8244522213935852, "rewards/total_composite/std": 0.14857427775859833, "sampling/importance_sampling_ratio/max": 1.8980882167816162, "sampling/importance_sampling_ratio/mean": 1.0049325227737427, "sampling/importance_sampling_ratio/min": 0.27536872029304504, "sampling/sampling_logp_difference/max": 1.2896442413330078, "sampling/sampling_logp_difference/mean": 0.031580884009599686, "step": 1885 }, { "clip_ratio/high_max": 0.020219704834744334, "clip_ratio/high_mean": 0.020219704834744334, "clip_ratio/low_mean": 0.00671045109629631, "clip_ratio/low_min": 0.00671045109629631, "clip_ratio/region_mean": 0.026930155931040645, "completions/clipped_ratio": 0.0, "completions/max_length": 222.0, "completions/max_terminated_length": 222.0, "completions/mean_length": 214.125, "completions/mean_terminated_length": 214.125, "completions/min_length": 204.0, "completions/min_terminated_length": 204.0, "entropy": 0.27855938114225864, "epoch": 0.07575209864642327, "frac_reward_zero_std": 0.0, "grad_norm": 2.436864137649536, "learning_rate": 4.287878787878788e-06, "loss": -0.0178, "num_tokens": 4253377.0, "reward": 0.8467762470245361, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_meter_mean": 0.9979674220085144, "reward_meter_std": 0.00041360704926773906, "reward_repeat_penalty_mean": 0.8818181753158569, "reward_repeat_penalty_std": 0.07008175551891327, "reward_std": 0.12215393036603928, "reward_total_composite_mean": 0.8467762470245361, "reward_total_composite_std": 0.12215393781661987, "reward_total_mean": 0.8467762470245361, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/meter/mean": 0.9979674220085144, "rewards/meter/std": 0.00041360704926773906, "rewards/repeat_penalty/mean": 0.8818181753158569, "rewards/repeat_penalty/std": 0.07008175551891327, "rewards/total_composite/mean": 0.8467762470245361, "rewards/total_composite/std": 0.12215393781661987, "sampling/importance_sampling_ratio/max": 1.72610342502594, "sampling/importance_sampling_ratio/mean": 1.0003207921981812, "sampling/importance_sampling_ratio/min": 0.11784233152866364, "sampling/sampling_logp_difference/max": 2.1384077072143555, "sampling/sampling_logp_difference/mean": 0.03404805809259415, "step": 1886 }, { "clip_ratio/high_max": 0.006850266130641103, "clip_ratio/high_mean": 0.006850266130641103, "clip_ratio/low_mean": 0.01478908269200474, "clip_ratio/low_min": 0.01478908269200474, "clip_ratio/region_mean": 0.021639348822645843, "completions/clipped_ratio": 0.0, "completions/max_length": 149.0, "completions/max_terminated_length": 149.0, "completions/mean_length": 144.625, "completions/mean_terminated_length": 144.625, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 0.25286005809903145, "epoch": 0.07579226412820822, "frac_reward_zero_std": 0.0, "grad_norm": 2.4621639251708984, "learning_rate": 4.284848484848485e-06, "loss": -0.0062, "num_tokens": 4255950.0, "reward": 0.8907864093780518, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997658908367157, "reward_meter_std": 0.0006960767204873264, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.10101525485515594, "reward_std": 0.10098133236169815, "reward_total_composite_mean": 0.8907864093780518, "reward_total_composite_std": 0.10098132491111755, "reward_total_mean": 0.8907864093780518, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997658908367157, "rewards/meter/std": 0.0006960767204873264, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.8907864093780518, "rewards/total_composite/std": 0.10098132491111755, "sampling/importance_sampling_ratio/max": 1.7592705488204956, "sampling/importance_sampling_ratio/mean": 1.0084306001663208, "sampling/importance_sampling_ratio/min": 0.29991504549980164, "sampling/sampling_logp_difference/max": 1.2042560577392578, "sampling/sampling_logp_difference/mean": 0.02850562147796154, "step": 1887 }, { "clip_ratio/high_max": 0.019573589437641203, "clip_ratio/high_mean": 0.019573589437641203, "clip_ratio/low_mean": 0.005528143374249339, "clip_ratio/low_min": 0.005528143374249339, "clip_ratio/region_mean": 0.025101732811890543, "completions/clipped_ratio": 0.0, "completions/max_length": 317.0, "completions/max_terminated_length": 317.0, "completions/mean_length": 296.625, "completions/mean_terminated_length": 296.625, "completions/min_length": 279.0, "completions/min_terminated_length": 279.0, "entropy": 0.28176482766866684, "epoch": 0.07583242960999317, "frac_reward_zero_std": 0.0, "grad_norm": 2.372718334197998, "learning_rate": 4.281818181818182e-06, "loss": -0.0096, "num_tokens": 4259851.0, "reward": 0.7886214256286621, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.90625, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.9935998320579529, "reward_meter_std": 0.005851938389241695, "reward_repeat_penalty_mean": 0.8760073184967041, "reward_repeat_penalty_std": 0.07234964519739151, "reward_std": 0.07851462066173553, "reward_total_composite_mean": 0.7886214256286621, "reward_total_composite_std": 0.07851462066173553, "reward_total_mean": 0.7886214256286621, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.90625, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.9935998320579529, "rewards/meter/std": 0.005851938389241695, "rewards/repeat_penalty/mean": 0.8760073184967041, "rewards/repeat_penalty/std": 0.07234964519739151, "rewards/total_composite/mean": 0.7886214256286621, "rewards/total_composite/std": 0.07851462066173553, "sampling/importance_sampling_ratio/max": 1.7818008661270142, "sampling/importance_sampling_ratio/mean": 1.0055367946624756, "sampling/importance_sampling_ratio/min": 0.08137635141611099, "sampling/sampling_logp_difference/max": 2.5086705684661865, "sampling/sampling_logp_difference/mean": 0.034045975655317307, "step": 1888 }, { "clip_ratio/high_max": 0.02012083982117474, "clip_ratio/high_mean": 0.02012083982117474, "clip_ratio/low_mean": 0.010531234613154083, "clip_ratio/low_min": 0.010531234613154083, "clip_ratio/region_mean": 0.030652074434328824, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 328.0, "completions/mean_terminated_length": 328.0, "completions/min_length": 305.0, "completions/min_terminated_length": 305.0, "entropy": 0.35988450422883034, "epoch": 0.07587259509177813, "frac_reward_zero_std": 0.0, "grad_norm": 1.960667610168457, "learning_rate": 4.278787878787879e-06, "loss": -0.0161, "num_tokens": 4264323.0, "reward": 0.5576076507568359, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6057692766189575, "reward_count_adherence_std": 0.027196412906050682, "reward_meter_mean": 0.9920902252197266, "reward_meter_std": 0.005891723092645407, "reward_repeat_penalty_mean": 0.9280506372451782, "reward_repeat_penalty_std": 0.05198076739907265, "reward_std": 0.03838847950100899, "reward_total_composite_mean": 0.5576076507568359, "reward_total_composite_std": 0.03838847950100899, "reward_total_mean": 0.5576076507568359, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6057692766189575, "rewards/count_adherence/std": 0.027196412906050682, "rewards/meter/mean": 0.9920902252197266, "rewards/meter/std": 0.005891723092645407, "rewards/repeat_penalty/mean": 0.9280506372451782, "rewards/repeat_penalty/std": 0.05198076739907265, "rewards/total_composite/mean": 0.5576076507568359, "rewards/total_composite/std": 0.03838847950100899, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0100011825561523, "sampling/importance_sampling_ratio/min": 0.0806482657790184, "sampling/sampling_logp_difference/max": 2.517657995223999, "sampling/sampling_logp_difference/mean": 0.03868624567985535, "step": 1889 }, { "clip_ratio/high_max": 0.003968254197388887, "clip_ratio/high_mean": 0.003968254197388887, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.006017434410750866, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 62.75, "completions/mean_terminated_length": 62.75, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.02277997531928122, "epoch": 0.07591276057356308, "frac_reward_zero_std": 0.0, "grad_norm": 1.555624008178711, "learning_rate": 4.275757575757576e-06, "loss": -0.0005, "num_tokens": 4266105.0, "reward": 0.9969385862350464, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969385862350464, "reward_meter_std": 0.00011288504174444824, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001128727599279955, "reward_total_composite_mean": 0.9969385862350464, "reward_total_composite_std": 0.00011288504174444824, "reward_total_mean": 0.9969385862350464, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969385862350464, "rewards/meter/std": 0.00011288504174444824, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969385862350464, "rewards/total_composite/std": 0.00011288504174444824, "sampling/importance_sampling_ratio/max": 1.6466243267059326, "sampling/importance_sampling_ratio/mean": 1.002489447593689, "sampling/importance_sampling_ratio/min": 0.7584096193313599, "sampling/sampling_logp_difference/max": 0.49872732162475586, "sampling/sampling_logp_difference/mean": 0.005147209390997887, "step": 1890 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.011020259640645236, "epoch": 0.07595292605534804, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.272727272727273e-06, "loss": 0.0, "num_tokens": 4267721.0, "reward": 0.9946010708808899, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946010708808899, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9946010708808899, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9946010708808899, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946010708808899, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946010708808899, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.095924735069275, "sampling/importance_sampling_ratio/mean": 1.0008796453475952, "sampling/importance_sampling_ratio/min": 0.9592980146408081, "sampling/sampling_logp_difference/max": 0.0915985256433487, "sampling/sampling_logp_difference/mean": 0.0011367781553417444, "step": 1891 }, { "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.01125222840346396, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.125, "completions/mean_terminated_length": 33.125, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.02548716589808464, "epoch": 0.07599309153713299, "frac_reward_zero_std": 0.0, "grad_norm": 4.663846969604492, "learning_rate": 4.2696969696969695e-06, "loss": 0.0082, "num_tokens": 4269442.0, "reward": 0.9981597661972046, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981597661972046, "reward_meter_std": 0.0018737530335783958, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0018737498903647065, "reward_total_composite_mean": 0.9981597661972046, "reward_total_composite_std": 0.0018737530335783958, "reward_total_mean": 0.9981597661972046, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981597661972046, "rewards/meter/std": 0.0018737530335783958, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981597661972046, "rewards/total_composite/std": 0.0018737530335783958, "sampling/importance_sampling_ratio/max": 1.9366061687469482, "sampling/importance_sampling_ratio/mean": 0.9993104338645935, "sampling/importance_sampling_ratio/min": 0.4537176191806793, "sampling/sampling_logp_difference/max": 0.7902803421020508, "sampling/sampling_logp_difference/mean": 0.011638238094747066, "step": 1892 }, { "clip_ratio/high_max": 0.043705128598958254, "clip_ratio/high_mean": 0.043705128598958254, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/region_mean": 0.048705128487199545, "completions/clipped_ratio": 0.0, "completions/max_length": 154.0, "completions/max_terminated_length": 154.0, "completions/mean_length": 147.875, "completions/mean_terminated_length": 147.875, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.3808891177177429, "epoch": 0.07603325701891794, "frac_reward_zero_std": 0.0, "grad_norm": 4.968896865844727, "learning_rate": 4.266666666666668e-06, "loss": 0.012, "num_tokens": 4272121.0, "reward": 0.9796580076217651, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997498095035553, "reward_meter_std": 0.0016895781736820936, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.049859896302223206, "reward_total_composite_mean": 0.9796580076217651, "reward_total_composite_std": 0.0498599074780941, "reward_total_mean": 0.9796580076217651, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997498095035553, "rewards/meter/std": 0.0016895781736820936, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9796580076217651, "rewards/total_composite/std": 0.0498599074780941, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002869725227356, "sampling/importance_sampling_ratio/min": 0.17087747156620026, "sampling/sampling_logp_difference/max": 1.7668085098266602, "sampling/sampling_logp_difference/mean": 0.05557761713862419, "step": 1893 }, { "clip_ratio/high_max": 0.03263740334659815, "clip_ratio/high_mean": 0.03263740334659815, "clip_ratio/low_mean": 0.012207416351884604, "clip_ratio/low_min": 0.012207416351884604, "clip_ratio/region_mean": 0.04484481969848275, "completions/clipped_ratio": 0.0, "completions/max_length": 189.0, "completions/max_terminated_length": 189.0, "completions/mean_length": 184.25, "completions/mean_terminated_length": 184.25, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.33961667120456696, "epoch": 0.0760734225007029, "frac_reward_zero_std": 0.0, "grad_norm": 3.8838560581207275, "learning_rate": 4.263636363636364e-06, "loss": 0.01, "num_tokens": 4274755.0, "reward": 0.9423543214797974, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978748559951782, "reward_meter_std": 0.0017333579016849399, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.08399210125207901, "reward_std": 0.08282525092363358, "reward_total_composite_mean": 0.9423543214797974, "reward_total_composite_std": 0.08282524347305298, "reward_total_mean": 0.9423543214797974, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978748559951782, "rewards/meter/std": 0.0017333579016849399, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.9423543214797974, "rewards/total_composite/std": 0.08282524347305298, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002802848815918, "sampling/importance_sampling_ratio/min": 0.11360271275043488, "sampling/sampling_logp_difference/max": 2.1750478744506836, "sampling/sampling_logp_difference/mean": 0.049167633056640625, "step": 1894 }, { "clip_ratio/high_max": 0.022966360847931355, "clip_ratio/high_mean": 0.022966360847931355, "clip_ratio/low_mean": 0.0031887756194919348, "clip_ratio/low_min": 0.0031887756194919348, "clip_ratio/region_mean": 0.02615513646742329, "completions/clipped_ratio": 0.0, "completions/max_length": 207.0, "completions/max_terminated_length": 207.0, "completions/mean_length": 200.75, "completions/mean_terminated_length": 200.75, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.3611125461757183, "epoch": 0.07611358798248785, "frac_reward_zero_std": 0.0, "grad_norm": 2.22912859916687, "learning_rate": 4.260606060606061e-06, "loss": -0.0041, "num_tokens": 4277761.0, "reward": 0.8416043519973755, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9934834837913513, "reward_meter_std": 0.003708732081577182, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.34394556283950806, "reward_total_composite_mean": 0.8416043519973755, "reward_total_composite_std": 0.34394556283950806, "reward_total_mean": 0.8416043519973755, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9934834837913513, "rewards/meter/std": 0.003708732081577182, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.8416043519973755, "rewards/total_composite/std": 0.34394556283950806, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0138887166976929, "sampling/importance_sampling_ratio/min": 0.2660071551799774, "sampling/sampling_logp_difference/max": 1.3242321014404297, "sampling/sampling_logp_difference/mean": 0.037548985332250595, "step": 1895 }, { "clip_ratio/high_max": 0.0007575757335871458, "clip_ratio/high_mean": 0.0007575757335871458, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0007575757335871458, "completions/clipped_ratio": 0.0, "completions/max_length": 165.0, "completions/max_terminated_length": 165.0, "completions/mean_length": 165.0, "completions/mean_terminated_length": 165.0, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.007689857040531933, "epoch": 0.0761537534642728, "frac_reward_zero_std": 0.0, "grad_norm": 0.0076941619627177715, "learning_rate": 4.2575757575757585e-06, "loss": 0.0005, "num_tokens": 4280665.0, "reward": 0.7771258354187012, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991617202758789, "reward_meter_std": 2.346204610148561e-06, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_std": 1.8390023797110189e-06, "reward_total_composite_mean": 0.7771258354187012, "reward_total_composite_std": 1.8245253841087106e-06, "reward_total_mean": 0.7771258354187012, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991617202758789, "rewards/meter/std": 2.346204610148561e-06, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7771258354187012, "rewards/total_composite/std": 1.8245253841087106e-06, "sampling/importance_sampling_ratio/max": 1.0792498588562012, "sampling/importance_sampling_ratio/mean": 1.0000821352005005, "sampling/importance_sampling_ratio/min": 0.25540316104888916, "sampling/sampling_logp_difference/max": 1.3649120330810547, "sampling/sampling_logp_difference/mean": 0.0018403534777462482, "step": 1896 }, { "clip_ratio/high_max": 0.04165800241753459, "clip_ratio/high_mean": 0.04165800241753459, "clip_ratio/low_mean": 0.0025167784187942743, "clip_ratio/low_min": 0.0025167784187942743, "clip_ratio/region_mean": 0.044174780836328864, "completions/clipped_ratio": 0.0, "completions/max_length": 158.0, "completions/max_terminated_length": 158.0, "completions/mean_length": 152.5, "completions/mean_terminated_length": 152.5, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.34401146695017815, "epoch": 0.07619391894605776, "frac_reward_zero_std": 0.0, "grad_norm": 4.121653079986572, "learning_rate": 4.254545454545455e-06, "loss": 0.0006, "num_tokens": 4283357.0, "reward": 0.9732140302658081, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9909997582435608, "reward_meter_std": 0.008123436011373997, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.04892229288816452, "reward_total_composite_mean": 0.9732140302658081, "reward_total_composite_std": 0.04892229288816452, "reward_total_mean": 0.9732140302658081, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9909997582435608, "rewards/meter/std": 0.008123436011373997, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9732140302658081, "rewards/total_composite/std": 0.04892229288816452, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006994605064392, "sampling/importance_sampling_ratio/min": 0.07180669158697128, "sampling/sampling_logp_difference/max": 2.633777618408203, "sampling/sampling_logp_difference/mean": 0.04408065602183342, "step": 1897 }, { "clip_ratio/high_max": 0.01466671982780099, "clip_ratio/high_mean": 0.01466671982780099, "clip_ratio/low_mean": 0.017732840729877353, "clip_ratio/low_min": 0.017732840729877353, "clip_ratio/region_mean": 0.03239956055767834, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 119.625, "completions/mean_terminated_length": 119.625, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.3594598062336445, "epoch": 0.07623408442784271, "frac_reward_zero_std": 0.0, "grad_norm": 2.702057361602783, "learning_rate": 4.251515151515152e-06, "loss": -0.001, "num_tokens": 4285658.0, "reward": 0.9938209652900696, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9938209652900696, "reward_meter_std": 0.002714785048738122, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0027147859800606966, "reward_total_composite_mean": 0.9938209652900696, "reward_total_composite_std": 0.002714785048738122, "reward_total_mean": 0.9938209652900696, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9938209652900696, "rewards/meter/std": 0.002714785048738122, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9938209652900696, "rewards/total_composite/std": 0.002714785048738122, "sampling/importance_sampling_ratio/max": 1.5793664455413818, "sampling/importance_sampling_ratio/mean": 1.0123997926712036, "sampling/importance_sampling_ratio/min": 0.3106164038181305, "sampling/sampling_logp_difference/max": 1.169196605682373, "sampling/sampling_logp_difference/mean": 0.039888497442007065, "step": 1898 }, { "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.002016128972172737, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.05075907311402261, "epoch": 0.07627424990962767, "frac_reward_zero_std": 0.0, "grad_norm": 5.739034175872803, "learning_rate": 4.248484848484849e-06, "loss": 0.0146, "num_tokens": 4287482.0, "reward": 0.9764682650566101, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9764682650566101, "reward_meter_std": 0.05796706676483154, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.057967059314250946, "reward_total_composite_mean": 0.9764682650566101, "reward_total_composite_std": 0.05796706676483154, "reward_total_mean": 0.9764682650566101, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9764682650566101, "rewards/meter/std": 0.05796706676483154, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9764682650566101, "rewards/total_composite/std": 0.05796706676483154, "sampling/importance_sampling_ratio/max": 1.515744924545288, "sampling/importance_sampling_ratio/mean": 0.9993007183074951, "sampling/importance_sampling_ratio/min": 0.33640727400779724, "sampling/sampling_logp_difference/max": 1.089432716369629, "sampling/sampling_logp_difference/mean": 0.010971495881676674, "step": 1899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 99.0, "completions/mean_terminated_length": 99.0, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.004198343900498003, "epoch": 0.07631441539141262, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.245454545454546e-06, "loss": 0.0, "num_tokens": 4289610.0, "reward": 0.9990620613098145, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990620613098145, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9990620613098145, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9990620613098145, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990620613098145, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990620613098145, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0143922567367554, "sampling/importance_sampling_ratio/mean": 1.00044846534729, "sampling/importance_sampling_ratio/min": 0.9927101135253906, "sampling/sampling_logp_difference/max": 0.014289772137999535, "sampling/sampling_logp_difference/mean": 0.0004978616489097476, "step": 1900 }, { "epoch": 0.07631441539141262, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 335.6923076923077, "eval_completions/max_terminated_length": 335.6923076923077, "eval_completions/mean_length": 199.51923076923077, "eval_completions/mean_terminated_length": 199.51923076923077, "eval_completions/min_length": 71.53846153846153, "eval_completions/min_terminated_length": 71.53846153846153, "eval_entropy": 0.15330308188612646, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 4289610.0, "eval_reward": 0.5190799373846787, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.8646397361388574, "eval_reward_count_adherence_std": 0.19298870517657354, "eval_reward_meter_mean": 0.7173460401021517, "eval_reward_meter_std": 0.4227825838785905, "eval_reward_repeat_penalty_mean": 0.8317501590802119, "eval_reward_repeat_penalty_std": 0.14178751237117326, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5190799373846787, "eval_reward_total_composite_std": 0.36580440631279576, "eval_reward_total_mean": 0.5190799373846787, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.8646397361388574, "eval_rewards/count_adherence/std": 0.19298870517657354, "eval_rewards/meter/mean": 0.7173460401021517, "eval_rewards/meter/std": 0.4227825838785905, "eval_rewards/repeat_penalty/mean": 0.8317501590802119, "eval_rewards/repeat_penalty/std": 0.14178751237117326, "eval_rewards/total_composite/mean": 0.5190799373846787, "eval_rewards/total_composite/std": 0.36580440631279576, "eval_runtime": 64.1724, "eval_samples_per_second": 1.621, "eval_sampling/importance_sampling_ratio/max": 1.388241483614995, "eval_sampling/importance_sampling_ratio/mean": 1.0033589509817271, "eval_sampling/importance_sampling_ratio/min": 0.3849313259124756, "eval_sampling/sampling_logp_difference/max": 0.989627324617826, "eval_sampling/sampling_logp_difference/mean": 0.014502167307700101, "eval_steps_per_second": 0.203, "step": 1900 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.013817269122228026, "epoch": 0.07635458087319758, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.242424242424243e-06, "loss": 0.0, "num_tokens": 4291354.0, "reward": 0.9946010708808899, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946010708808899, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9946010708808899, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9946010708808899, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946010708808899, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946010708808899, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0319656133651733, "sampling/importance_sampling_ratio/mean": 1.00002121925354, "sampling/importance_sampling_ratio/min": 0.8426884412765503, "sampling/sampling_logp_difference/max": 0.17115801572799683, "sampling/sampling_logp_difference/mean": 0.0012767898151651025, "step": 1901 }, { "clip_ratio/high_max": 0.005102040711790323, "clip_ratio/high_mean": 0.005102040711790323, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005102040711790323, "completions/clipped_ratio": 0.0, "completions/max_length": 51.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 50.75, "completions/mean_terminated_length": 50.75, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.0905577321536839, "epoch": 0.07639474635498253, "frac_reward_zero_std": 0.0, "grad_norm": 5.961615085601807, "learning_rate": 4.2393939393939395e-06, "loss": 0.0286, "num_tokens": 4293032.0, "reward": 0.9352545142173767, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9352545142173767, "reward_meter_std": 0.01202213205397129, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.012022126466035843, "reward_total_composite_mean": 0.9352545142173767, "reward_total_composite_std": 0.01202213205397129, "reward_total_mean": 0.9352545142173767, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9352545142173767, "rewards/meter/std": 0.01202213205397129, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9352545142173767, "rewards/total_composite/std": 0.01202213205397129, "sampling/importance_sampling_ratio/max": 1.7655236721038818, "sampling/importance_sampling_ratio/mean": 0.9972783327102661, "sampling/importance_sampling_ratio/min": 0.3693333864212036, "sampling/sampling_logp_difference/max": 0.9960556030273438, "sampling/sampling_logp_difference/mean": 0.016524022445082664, "step": 1902 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 127.0, "completions/max_terminated_length": 127.0, "completions/mean_length": 127.0, "completions/mean_terminated_length": 127.0, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.008461171586532146, "epoch": 0.07643491183676748, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.236363636363637e-06, "loss": 0.0, "num_tokens": 4295672.0, "reward": 0.8549612164497375, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974547624588013, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.8549612164497375, "reward_total_composite_std": 0.0, "reward_total_mean": 0.8549612164497375, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974547624588013, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8549612164497375, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0233416557312012, "sampling/importance_sampling_ratio/mean": 1.0008466243743896, "sampling/importance_sampling_ratio/min": 0.9878833293914795, "sampling/sampling_logp_difference/max": 0.02307342365384102, "sampling/sampling_logp_difference/mean": 0.000930364360101521, "step": 1903 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.006679805053863674, "epoch": 0.07647507731855244, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.233333333333334e-06, "loss": 0.0, "num_tokens": 4297480.0, "reward": 0.997582197189331, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997582197189331, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.997582197189331, "reward_total_composite_std": 0.0, "reward_total_mean": 0.997582197189331, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997582197189331, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997582197189331, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0086549520492554, "sampling/importance_sampling_ratio/mean": 1.0005828142166138, "sampling/importance_sampling_ratio/min": 0.990973711013794, "sampling/sampling_logp_difference/max": 0.00906725786626339, "sampling/sampling_logp_difference/mean": 0.000656367396004498, "step": 1904 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0026041667442768812, "clip_ratio/low_min": 0.0026041667442768812, "clip_ratio/region_mean": 0.0026041667442768812, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 95.125, "completions/mean_terminated_length": 95.125, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.03163000999484211, "epoch": 0.07651524280033739, "frac_reward_zero_std": 0.0, "grad_norm": 2.374253034591675, "learning_rate": 4.2303030303030304e-06, "loss": 0.0031, "num_tokens": 4299561.0, "reward": 0.9972648620605469, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972648620605469, "reward_meter_std": 0.00012022549344692379, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00012023184535792097, "reward_total_composite_mean": 0.9972648620605469, "reward_total_composite_std": 0.00012022549344692379, "reward_total_mean": 0.9972648620605469, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972648620605469, "rewards/meter/std": 0.00012022549344692379, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972648620605469, "rewards/total_composite/std": 0.00012022549344692379, "sampling/importance_sampling_ratio/max": 1.5949854850769043, "sampling/importance_sampling_ratio/mean": 1.0006614923477173, "sampling/importance_sampling_ratio/min": 0.34241965413093567, "sampling/sampling_logp_difference/max": 1.0717182159423828, "sampling/sampling_logp_difference/mean": 0.008180440403521061, "step": 1905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.0030257631733547896, "epoch": 0.07655540828212234, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.227272727272728e-06, "loss": 0.0, "num_tokens": 4301393.0, "reward": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998939037322998, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "reward_total_mean": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998939037322998, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0110995769500732, "sampling/importance_sampling_ratio/mean": 1.0002886056900024, "sampling/importance_sampling_ratio/min": 0.9955568313598633, "sampling/sampling_logp_difference/max": 0.011038463562726974, "sampling/sampling_logp_difference/mean": 0.0003236948396079242, "step": 1906 }, { "clip_ratio/high_max": 0.014517532312311232, "clip_ratio/high_mean": 0.014517532312311232, "clip_ratio/low_mean": 0.0019704491132870317, "clip_ratio/low_min": 0.0019704491132870317, "clip_ratio/region_mean": 0.016487981425598264, "completions/clipped_ratio": 0.0, "completions/max_length": 258.0, "completions/max_terminated_length": 258.0, "completions/mean_length": 251.625, "completions/mean_terminated_length": 251.625, "completions/min_length": 243.0, "completions/min_terminated_length": 243.0, "entropy": 0.17152301780879498, "epoch": 0.0765955737639073, "frac_reward_zero_std": 0.0, "grad_norm": 2.0788662433624268, "learning_rate": 4.224242424242425e-06, "loss": 0.0114, "num_tokens": 4305238.0, "reward": 0.7676311731338501, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997933030128479, "reward_meter_std": 0.00043991892016492784, "reward_repeat_penalty_mean": 0.7692307829856873, "reward_repeat_penalty_std": 0.07121692597866058, "reward_std": 0.07094840705394745, "reward_total_composite_mean": 0.7676311731338501, "reward_total_composite_std": 0.07094842940568924, "reward_total_mean": 0.7676311731338501, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997933030128479, "rewards/meter/std": 0.00043991892016492784, "rewards/repeat_penalty/mean": 0.7692307829856873, "rewards/repeat_penalty/std": 0.07121692597866058, "rewards/total_composite/mean": 0.7676311731338501, "rewards/total_composite/std": 0.07094842940568924, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0044364929199219, "sampling/importance_sampling_ratio/min": 0.3728809952735901, "sampling/sampling_logp_difference/max": 0.9864959716796875, "sampling/sampling_logp_difference/mean": 0.02055802382528782, "step": 1907 }, { "clip_ratio/high_max": 0.013561643892899156, "clip_ratio/high_mean": 0.013561643892899156, "clip_ratio/low_mean": 0.01705764839425683, "clip_ratio/low_min": 0.01705764839425683, "clip_ratio/region_mean": 0.030619292287155986, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 73.875, "completions/mean_terminated_length": 73.875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.2469378486275673, "epoch": 0.07663573924569225, "frac_reward_zero_std": 0.0, "grad_norm": 4.272599697113037, "learning_rate": 4.221212121212121e-06, "loss": 0.0026, "num_tokens": 4307205.0, "reward": 0.9967119693756104, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9967119693756104, "reward_meter_std": 0.001478171325288713, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0014781749341636896, "reward_total_composite_mean": 0.9967119693756104, "reward_total_composite_std": 0.001478171325288713, "reward_total_mean": 0.9967119693756104, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9967119693756104, "rewards/meter/std": 0.001478171325288713, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967119693756104, "rewards/total_composite/std": 0.001478171325288713, "sampling/importance_sampling_ratio/max": 1.7452210187911987, "sampling/importance_sampling_ratio/mean": 1.0084983110427856, "sampling/importance_sampling_ratio/min": 0.2773742079734802, "sampling/sampling_logp_difference/max": 1.2823877334594727, "sampling/sampling_logp_difference/mean": 0.034489989280700684, "step": 1908 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0012135922443121672, "clip_ratio/low_min": 0.0012135922443121672, "clip_ratio/region_mean": 0.0012135922443121672, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 103.0, "completions/mean_terminated_length": 103.0, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.009892041911371052, "epoch": 0.0766759047274772, "frac_reward_zero_std": 0.0, "grad_norm": 0.519289493560791, "learning_rate": 4.218181818181819e-06, "loss": -0.0005, "num_tokens": 4309357.0, "reward": 0.9975078105926514, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975078105926514, "reward_meter_std": 3.284694321337156e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.283110709162429e-05, "reward_total_composite_mean": 0.9975078105926514, "reward_total_composite_std": 3.284694321337156e-05, "reward_total_mean": 0.9975078105926514, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975078105926514, "rewards/meter/std": 3.284694321337156e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975078105926514, "rewards/total_composite/std": 3.284694321337156e-05, "sampling/importance_sampling_ratio/max": 1.714148759841919, "sampling/importance_sampling_ratio/mean": 1.0008713006973267, "sampling/importance_sampling_ratio/min": 0.5307995080947876, "sampling/sampling_logp_difference/max": 0.6333708763122559, "sampling/sampling_logp_difference/mean": 0.0023622734006494284, "step": 1909 }, { "clip_ratio/high_max": 0.0004071661096531898, "clip_ratio/high_mean": 0.0004071661096531898, "clip_ratio/low_mean": 0.006014678278006613, "clip_ratio/low_min": 0.006014678278006613, "clip_ratio/region_mean": 0.006421844387659803, "completions/clipped_ratio": 0.0, "completions/max_length": 307.0, "completions/max_terminated_length": 307.0, "completions/mean_length": 291.125, "completions/mean_terminated_length": 291.125, "completions/min_length": 258.0, "completions/min_terminated_length": 258.0, "entropy": 0.0457235153298825, "epoch": 0.07671607020926216, "frac_reward_zero_std": 0.0, "grad_norm": 1.9287523031234741, "learning_rate": 4.215151515151515e-06, "loss": -0.0326, "num_tokens": 4313350.0, "reward": 0.6118792295455933, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.059391383081674576, "reward_meter_mean": 0.9932549595832825, "reward_meter_std": 0.0015650950372219086, "reward_repeat_penalty_mean": 0.6931459903717041, "reward_repeat_penalty_std": 0.010692903771996498, "reward_std": 0.04135050252079964, "reward_total_composite_mean": 0.6118792295455933, "reward_total_composite_std": 0.041350506246089935, "reward_total_mean": 0.6118792295455933, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.059391383081674576, "rewards/meter/mean": 0.9932549595832825, "rewards/meter/std": 0.0015650950372219086, "rewards/repeat_penalty/mean": 0.6931459903717041, "rewards/repeat_penalty/std": 0.010692903771996498, "rewards/total_composite/mean": 0.6118792295455933, "rewards/total_composite/std": 0.041350506246089935, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9992997646331787, "sampling/importance_sampling_ratio/min": 0.1384636014699936, "sampling/sampling_logp_difference/max": 1.9771476984024048, "sampling/sampling_logp_difference/mean": 0.010057836771011353, "step": 1910 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.004040350759169087, "epoch": 0.07675623569104711, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.212121212121212e-06, "loss": 0.0, "num_tokens": 4315262.0, "reward": 0.998939037322998, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998939037322998, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.998939037322998, "reward_total_composite_std": 0.0, "reward_total_mean": 0.998939037322998, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998939037322998, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998939037322998, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.006282091140747, "sampling/importance_sampling_ratio/mean": 1.0003656148910522, "sampling/importance_sampling_ratio/min": 0.9994460940361023, "sampling/sampling_logp_difference/max": 0.006262333132326603, "sampling/sampling_logp_difference/mean": 0.0003693441394716501, "step": 1911 }, { "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018939394503831863, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.011784833506681025, "epoch": 0.07679640117283207, "frac_reward_zero_std": 0.0, "grad_norm": 0.2683267891407013, "learning_rate": 4.2090909090909095e-06, "loss": 0.0001, "num_tokens": 4317134.0, "reward": 0.9989374279975891, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989374279975891, "reward_meter_std": 4.551860001811292e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.551859547063941e-06, "reward_total_composite_mean": 0.9989374279975891, "reward_total_composite_std": 4.551860001811292e-06, "reward_total_mean": 0.9989374279975891, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989374279975891, "rewards/meter/std": 4.551860001811292e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989374279975891, "rewards/total_composite/std": 4.551860001811292e-06, "sampling/importance_sampling_ratio/max": 1.2207425832748413, "sampling/importance_sampling_ratio/mean": 1.0016252994537354, "sampling/importance_sampling_ratio/min": 0.8604245185852051, "sampling/sampling_logp_difference/max": 0.19945931434631348, "sampling/sampling_logp_difference/mean": 0.002090727211907506, "step": 1912 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.018943659961223602, "epoch": 0.07683656665461702, "frac_reward_zero_std": 0.0, "grad_norm": 0.7913539409637451, "learning_rate": 4.206060606060606e-06, "loss": -0.0002, "num_tokens": 4318894.0, "reward": 0.9945913553237915, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9945913553237915, "reward_meter_std": 2.7374408091418445e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.7371370379114524e-05, "reward_total_composite_mean": 0.9945913553237915, "reward_total_composite_std": 2.7374408091418445e-05, "reward_total_mean": 0.9945913553237915, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9945913553237915, "rewards/meter/std": 2.7374408091418445e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9945913553237915, "rewards/total_composite/std": 2.7374408091418445e-05, "sampling/importance_sampling_ratio/max": 1.2506028413772583, "sampling/importance_sampling_ratio/mean": 1.0011041164398193, "sampling/importance_sampling_ratio/min": 0.8597717881202698, "sampling/sampling_logp_difference/max": 0.22362565994262695, "sampling/sampling_logp_difference/mean": 0.003088480094447732, "step": 1913 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.005322357552358881, "epoch": 0.07687673213640198, "frac_reward_zero_std": 0.0, "grad_norm": 0.36171841621398926, "learning_rate": 4.203030303030303e-06, "loss": -0.0005, "num_tokens": 4320646.0, "reward": 0.9989273548126221, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989273548126221, "reward_meter_std": 3.3127438655355945e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.312742410344072e-05, "reward_total_composite_mean": 0.9989273548126221, "reward_total_composite_std": 3.3127438655355945e-05, "reward_total_mean": 0.9989273548126221, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989273548126221, "rewards/meter/std": 3.3127438655355945e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989273548126221, "rewards/total_composite/std": 3.3127438655355945e-05, "sampling/importance_sampling_ratio/max": 1.0103963613510132, "sampling/importance_sampling_ratio/mean": 0.9987910985946655, "sampling/importance_sampling_ratio/min": 0.14600348472595215, "sampling/sampling_logp_difference/max": 1.9241247177124023, "sampling/sampling_logp_difference/mean": 0.004210924729704857, "step": 1914 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.0010245901066809893, "clip_ratio/low_min": 0.0010245901066809893, "clip_ratio/region_mean": 0.003073770320042968, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 122.0, "completions/mean_terminated_length": 122.0, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.02201851806603372, "epoch": 0.07691689761818693, "frac_reward_zero_std": 0.0, "grad_norm": 0.24196073412895203, "learning_rate": 4.2000000000000004e-06, "loss": -0.0005, "num_tokens": 4323030.0, "reward": 0.8535913228988647, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.995856523513794, "reward_meter_std": 2.842021240212489e-05, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 2.4356924768653698e-05, "reward_total_composite_mean": 0.8535913228988647, "reward_total_composite_std": 2.4363118427572772e-05, "reward_total_mean": 0.8535913228988647, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.995856523513794, "rewards/meter/std": 2.842021240212489e-05, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8535913228988647, "rewards/total_composite/std": 2.4363118427572772e-05, "sampling/importance_sampling_ratio/max": 1.361249566078186, "sampling/importance_sampling_ratio/mean": 1.0021631717681885, "sampling/importance_sampling_ratio/min": 0.7478480339050293, "sampling/sampling_logp_difference/max": 0.30840301513671875, "sampling/sampling_logp_difference/mean": 0.00429805601015687, "step": 1915 }, { "clip_ratio/high_max": 0.005000000121071935, "clip_ratio/high_mean": 0.005000000121071935, "clip_ratio/low_mean": 0.00856418942566961, "clip_ratio/low_min": 0.00856418942566961, "clip_ratio/region_mean": 0.013564189546741545, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 74.25, "completions/mean_terminated_length": 74.25, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.18484536185860634, "epoch": 0.07695706309997188, "frac_reward_zero_std": 0.0, "grad_norm": 2.755591630935669, "learning_rate": 4.196969696969697e-06, "loss": 0.0035, "num_tokens": 4324864.0, "reward": 0.9977705478668213, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977705478668213, "reward_meter_std": 0.0007801271858625114, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007801253814250231, "reward_total_composite_mean": 0.9977705478668213, "reward_total_composite_std": 0.0007801271858625114, "reward_total_mean": 0.9977705478668213, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977705478668213, "rewards/meter/std": 0.0007801271858625114, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977705478668213, "rewards/total_composite/std": 0.0007801271858625114, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0088934898376465, "sampling/importance_sampling_ratio/min": 0.35886335372924805, "sampling/sampling_logp_difference/max": 1.0248136520385742, "sampling/sampling_logp_difference/mean": 0.02434798702597618, "step": 1916 }, { "clip_ratio/high_max": 0.038703347789123654, "clip_ratio/high_mean": 0.038703347789123654, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/region_mean": 0.045460104709491134, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 81.5, "completions/mean_terminated_length": 81.5, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.30126412957906723, "epoch": 0.07699722858175684, "frac_reward_zero_std": 0.0, "grad_norm": 7.223612308502197, "learning_rate": 4.193939393939394e-06, "loss": 0.1368, "num_tokens": 4326780.0, "reward": 0.9299638271331787, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9922255277633667, "reward_meter_std": 0.006029736250638962, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.17459867894649506, "reward_total_composite_mean": 0.9299638271331787, "reward_total_composite_std": 0.17459870874881744, "reward_total_mean": 0.9299638271331787, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9922255277633667, "rewards/meter/std": 0.006029736250638962, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9299638271331787, "rewards/total_composite/std": 0.17459870874881744, "sampling/importance_sampling_ratio/max": 1.5985634326934814, "sampling/importance_sampling_ratio/mean": 1.0046718120574951, "sampling/importance_sampling_ratio/min": 0.02325153723359108, "sampling/sampling_logp_difference/max": 3.7613840103149414, "sampling/sampling_logp_difference/mean": 0.043613094836473465, "step": 1917 }, { "clip_ratio/high_max": 0.017437846050597727, "clip_ratio/high_mean": 0.017437846050597727, "clip_ratio/low_mean": 0.005113846738822758, "clip_ratio/low_min": 0.005113846738822758, "clip_ratio/region_mean": 0.022551692789420485, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 72.5, "completions/mean_terminated_length": 72.5, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.15394389163702726, "epoch": 0.07703739406354179, "frac_reward_zero_std": 0.0, "grad_norm": 3.7035961151123047, "learning_rate": 4.190909090909091e-06, "loss": 0.0163, "num_tokens": 4328592.0, "reward": 0.986924946308136, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.986924946308136, "reward_meter_std": 0.002686952007934451, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0026869599241763353, "reward_total_composite_mean": 0.986924946308136, "reward_total_composite_std": 0.002686952007934451, "reward_total_mean": 0.986924946308136, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.986924946308136, "rewards/meter/std": 0.002686952007934451, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.986924946308136, "rewards/total_composite/std": 0.002686952007934451, "sampling/importance_sampling_ratio/max": 1.7578001022338867, "sampling/importance_sampling_ratio/mean": 1.0001866817474365, "sampling/importance_sampling_ratio/min": 0.46272483468055725, "sampling/sampling_logp_difference/max": 0.770622730255127, "sampling/sampling_logp_difference/mean": 0.016851795837283134, "step": 1918 }, { "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005434782709926367, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.027878023218363523, "epoch": 0.07707755954532675, "frac_reward_zero_std": 0.0, "grad_norm": 1.28860342502594, "learning_rate": 4.187878787878788e-06, "loss": -0.0004, "num_tokens": 4330512.0, "reward": 0.9976222515106201, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976222515106201, "reward_meter_std": 4.285174509277567e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.2851730540860444e-05, "reward_total_composite_mean": 0.9976222515106201, "reward_total_composite_std": 4.285174509277567e-05, "reward_total_mean": 0.9976222515106201, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976222515106201, "rewards/meter/std": 4.285174509277567e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976222515106201, "rewards/total_composite/std": 4.285174509277567e-05, "sampling/importance_sampling_ratio/max": 1.3529547452926636, "sampling/importance_sampling_ratio/mean": 0.9985252022743225, "sampling/importance_sampling_ratio/min": 0.4213259816169739, "sampling/sampling_logp_difference/max": 0.8643484711647034, "sampling/sampling_logp_difference/mean": 0.006954767741262913, "step": 1919 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 131.875, "completions/mean_terminated_length": 131.875, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.01177009951788932, "epoch": 0.0771177250271117, "frac_reward_zero_std": 0.0, "grad_norm": 0.23397041857242584, "learning_rate": 4.184848484848485e-06, "loss": -0.0001, "num_tokens": 4332959.0, "reward": 0.8563849925994873, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991158246994019, "reward_meter_std": 1.4611580809287261e-05, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 1.2510253327491228e-05, "reward_total_composite_mean": 0.8563849925994873, "reward_total_composite_std": 1.2507220162660815e-05, "reward_total_mean": 0.8563849925994873, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991158246994019, "rewards/meter/std": 1.4611580809287261e-05, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8563849925994873, "rewards/total_composite/std": 1.2507220162660815e-05, "sampling/importance_sampling_ratio/max": 1.0698660612106323, "sampling/importance_sampling_ratio/mean": 0.9987531304359436, "sampling/importance_sampling_ratio/min": 0.2013085037469864, "sampling/sampling_logp_difference/max": 1.6029167175292969, "sampling/sampling_logp_difference/mean": 0.004140977747738361, "step": 1920 }, { "clip_ratio/high_max": 0.02495911391451955, "clip_ratio/high_mean": 0.02495911391451955, "clip_ratio/low_mean": 0.011023133061826229, "clip_ratio/low_min": 0.011023133061826229, "clip_ratio/region_mean": 0.03598224697634578, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 114.25, "completions/mean_terminated_length": 114.25, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.3176443073898554, "epoch": 0.07715789050889665, "frac_reward_zero_std": 0.0, "grad_norm": 3.0183725357055664, "learning_rate": 4.181818181818182e-06, "loss": 0.0041, "num_tokens": 4335289.0, "reward": 0.9916590452194214, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9916590452194214, "reward_meter_std": 0.004383288789540529, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004383287392556667, "reward_total_composite_mean": 0.9916590452194214, "reward_total_composite_std": 0.004383288789540529, "reward_total_mean": 0.9916590452194214, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9916590452194214, "rewards/meter/std": 0.004383288789540529, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9916590452194214, "rewards/total_composite/std": 0.004383288789540529, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0076040029525757, "sampling/importance_sampling_ratio/min": 0.4941423237323761, "sampling/sampling_logp_difference/max": 0.9611685276031494, "sampling/sampling_logp_difference/mean": 0.03317651152610779, "step": 1921 }, { "clip_ratio/high_max": 0.001923076924867928, "clip_ratio/high_mean": 0.001923076924867928, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.001923076924867928, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.015987101825885475, "epoch": 0.07719805599068161, "frac_reward_zero_std": 0.0, "grad_norm": 0.0710740014910698, "learning_rate": 4.1787878787878795e-06, "loss": 0.0006, "num_tokens": 4337264.0, "reward": 0.9989427924156189, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989427924156189, "reward_meter_std": 1.0663165085134096e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.0657141501724254e-05, "reward_total_composite_mean": 0.9989427924156189, "reward_total_composite_std": 1.0663165085134096e-05, "reward_total_mean": 0.9989427924156189, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989427924156189, "rewards/meter/std": 1.0663165085134096e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989427924156189, "rewards/total_composite/std": 1.0663165085134096e-05, "sampling/importance_sampling_ratio/max": 1.1512835025787354, "sampling/importance_sampling_ratio/mean": 0.9995160102844238, "sampling/importance_sampling_ratio/min": 0.0777566209435463, "sampling/sampling_logp_difference/max": 2.554171562194824, "sampling/sampling_logp_difference/mean": 0.006586765870451927, "step": 1922 }, { "clip_ratio/high_max": 0.0031607006676495075, "clip_ratio/high_mean": 0.0031607006676495075, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0031607006676495075, "completions/clipped_ratio": 0.0, "completions/max_length": 202.0, "completions/max_terminated_length": 202.0, "completions/mean_length": 201.125, "completions/mean_terminated_length": 201.125, "completions/min_length": 195.0, "completions/min_terminated_length": 195.0, "entropy": 0.01974188955500722, "epoch": 0.07723822147246656, "frac_reward_zero_std": 0.0, "grad_norm": 0.6413597464561462, "learning_rate": 4.175757575757576e-06, "loss": 0.0104, "num_tokens": 4340257.0, "reward": 0.5811127424240112, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961963891983032, "reward_meter_std": 7.053603621898219e-05, "reward_repeat_penalty_mean": 0.7291666865348816, "reward_repeat_penalty_std": 0.058925554156303406, "reward_std": 0.0469353049993515, "reward_total_composite_mean": 0.5811127424240112, "reward_total_composite_std": 0.0469353049993515, "reward_total_mean": 0.5811127424240112, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961963891983032, "rewards/meter/std": 7.053603621898219e-05, "rewards/repeat_penalty/mean": 0.7291666865348816, "rewards/repeat_penalty/std": 0.058925554156303406, "rewards/total_composite/mean": 0.5811127424240112, "rewards/total_composite/std": 0.0469353049993515, "sampling/importance_sampling_ratio/max": 1.3452818393707275, "sampling/importance_sampling_ratio/mean": 0.9994967579841614, "sampling/importance_sampling_ratio/min": 0.3703322112560272, "sampling/sampling_logp_difference/max": 0.9933547973632812, "sampling/sampling_logp_difference/mean": 0.00489839119836688, "step": 1923 }, { "clip_ratio/high_max": 0.040673923096619546, "clip_ratio/high_mean": 0.040673923096619546, "clip_ratio/low_mean": 0.004587155766785145, "clip_ratio/low_min": 0.004587155766785145, "clip_ratio/region_mean": 0.04526107886340469, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 80.75, "completions/mean_terminated_length": 80.75, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.3375435955822468, "epoch": 0.07727838695425152, "frac_reward_zero_std": 0.0, "grad_norm": 6.834925174713135, "learning_rate": 4.172727272727273e-06, "loss": 0.1384, "num_tokens": 4342199.0, "reward": 0.928816556930542, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.9895592927932739, "reward_meter_std": 0.01372271403670311, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.17933164536952972, "reward_total_composite_mean": 0.928816556930542, "reward_total_composite_std": 0.17933166027069092, "reward_total_mean": 0.928816556930542, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.9895592927932739, "rewards/meter/std": 0.01372271403670311, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.928816556930542, "rewards/total_composite/std": 0.17933166027069092, "sampling/importance_sampling_ratio/max": 1.9494916200637817, "sampling/importance_sampling_ratio/mean": 1.009357213973999, "sampling/importance_sampling_ratio/min": 0.015856586396694183, "sampling/sampling_logp_difference/max": 4.14417028427124, "sampling/sampling_logp_difference/mean": 0.048581015318632126, "step": 1924 }, { "clip_ratio/high_max": 0.004310344811528921, "clip_ratio/high_mean": 0.004310344811528921, "clip_ratio/low_mean": 0.004387315362691879, "clip_ratio/low_min": 0.004387315362691879, "clip_ratio/region_mean": 0.0086976601742208, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 57.75, "completions/mean_terminated_length": 57.75, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.025414136587642133, "epoch": 0.07731855243603647, "frac_reward_zero_std": 0.0, "grad_norm": 16.744361877441406, "learning_rate": 4.1696969696969705e-06, "loss": 0.0103, "num_tokens": 4343965.0, "reward": 0.9946485757827759, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946485757827759, "reward_meter_std": 0.00013940427743364125, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001394047576468438, "reward_total_composite_mean": 0.9946485757827759, "reward_total_composite_std": 0.00013940427743364125, "reward_total_mean": 0.9946485757827759, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946485757827759, "rewards/meter/std": 0.00013940427743364125, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946485757827759, "rewards/total_composite/std": 0.00013940427743364125, "sampling/importance_sampling_ratio/max": 1.1374722719192505, "sampling/importance_sampling_ratio/mean": 0.9992939233779907, "sampling/importance_sampling_ratio/min": 0.6150954961776733, "sampling/sampling_logp_difference/max": 0.48597773909568787, "sampling/sampling_logp_difference/mean": 0.004949766676872969, "step": 1925 }, { "clip_ratio/high_max": 0.012181702419184148, "clip_ratio/high_mean": 0.012181702419184148, "clip_ratio/low_mean": 0.003884511475916952, "clip_ratio/low_min": 0.003884511475916952, "clip_ratio/region_mean": 0.0160662138951011, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 297.0, "completions/mean_terminated_length": 297.0, "completions/min_length": 282.0, "completions/min_terminated_length": 282.0, "entropy": 0.12555130943655968, "epoch": 0.07735871791782142, "frac_reward_zero_std": 0.0, "grad_norm": 2.593313694000244, "learning_rate": 4.166666666666667e-06, "loss": 0.0014, "num_tokens": 4347997.0, "reward": 0.661956250667572, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.9778913855552673, "reward_meter_std": 0.025907516479492188, "reward_repeat_penalty_mean": 0.7741815447807312, "reward_repeat_penalty_std": 0.09523647278547287, "reward_std": 0.08313874155282974, "reward_total_composite_mean": 0.661956250667572, "reward_total_composite_std": 0.08313874155282974, "reward_total_mean": 0.661956250667572, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.9778913855552673, "rewards/meter/std": 0.025907516479492188, "rewards/repeat_penalty/mean": 0.7741815447807312, "rewards/repeat_penalty/std": 0.09523647278547287, "rewards/total_composite/mean": 0.661956250667572, "rewards/total_composite/std": 0.08313874155282974, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0018694400787354, "sampling/importance_sampling_ratio/min": 0.0005214267293922603, "sampling/sampling_logp_difference/max": 7.558941841125488, "sampling/sampling_logp_difference/mean": 0.027429981157183647, "step": 1926 }, { "clip_ratio/high_max": 0.0025252525229007006, "clip_ratio/high_mean": 0.0025252525229007006, "clip_ratio/low_mean": 0.0025252525229007006, "clip_ratio/low_min": 0.0025252525229007006, "clip_ratio/region_mean": 0.005050505045801401, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 98.875, "completions/mean_terminated_length": 98.875, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.02114183083176613, "epoch": 0.07739888339960638, "frac_reward_zero_std": 0.0, "grad_norm": 0.516058087348938, "learning_rate": 4.163636363636364e-06, "loss": -0.0006, "num_tokens": 4350132.0, "reward": 0.9990072250366211, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990072250366211, "reward_meter_std": 0.00012045173207297921, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00012044955656165257, "reward_total_composite_mean": 0.9990072250366211, "reward_total_composite_std": 0.00012045173207297921, "reward_total_mean": 0.9990072250366211, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990072250366211, "rewards/meter/std": 0.00012045173207297921, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990072250366211, "rewards/total_composite/std": 0.00012045173207297921, "sampling/importance_sampling_ratio/max": 1.2373576164245605, "sampling/importance_sampling_ratio/mean": 0.9992603659629822, "sampling/importance_sampling_ratio/min": 0.5991997718811035, "sampling/sampling_logp_difference/max": 0.5121603012084961, "sampling/sampling_logp_difference/mean": 0.004538480192422867, "step": 1927 }, { "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/low_mean": 0.006048386916518211, "clip_ratio/low_min": 0.006048386916518211, "clip_ratio/region_mean": 0.008032514015212655, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 61.375, "completions/mean_terminated_length": 61.375, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.04211301123723388, "epoch": 0.07743904888139133, "frac_reward_zero_std": 0.0, "grad_norm": 3.781970977783203, "learning_rate": 4.160606060606061e-06, "loss": -0.0147, "num_tokens": 4351735.0, "reward": 0.9964951276779175, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9964951276779175, "reward_meter_std": 0.00043850651127286255, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004385166976135224, "reward_total_composite_mean": 0.9964951276779175, "reward_total_composite_std": 0.00043850651127286255, "reward_total_mean": 0.9964951276779175, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9964951276779175, "rewards/meter/std": 0.00043850651127286255, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964951276779175, "rewards/total_composite/std": 0.00043850651127286255, "sampling/importance_sampling_ratio/max": 1.4405137300491333, "sampling/importance_sampling_ratio/mean": 1.0027474164962769, "sampling/importance_sampling_ratio/min": 0.6272673606872559, "sampling/sampling_logp_difference/max": 0.4663825035095215, "sampling/sampling_logp_difference/mean": 0.007140871603041887, "step": 1928 }, { "clip_ratio/high_max": 0.0013888889225199819, "clip_ratio/high_mean": 0.0013888889225199819, "clip_ratio/low_mean": 0.0126003761542961, "clip_ratio/low_min": 0.0126003761542961, "clip_ratio/region_mean": 0.013989265076816082, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 81.875, "completions/mean_terminated_length": 81.875, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.07624188321642578, "epoch": 0.07747921436317629, "frac_reward_zero_std": 0.0, "grad_norm": 3.9287514686584473, "learning_rate": 4.157575757575758e-06, "loss": -0.0459, "num_tokens": 4353662.0, "reward": 0.9926681518554688, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9926681518554688, "reward_meter_std": 0.00178977579344064, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0017897515790537, "reward_total_composite_mean": 0.9926681518554688, "reward_total_composite_std": 0.00178977579344064, "reward_total_mean": 0.9926681518554688, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9926681518554688, "rewards/meter/std": 0.00178977579344064, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926681518554688, "rewards/total_composite/std": 0.00178977579344064, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0027321577072144, "sampling/importance_sampling_ratio/min": 0.38019606471061707, "sampling/sampling_logp_difference/max": 0.9670681953430176, "sampling/sampling_logp_difference/mean": 0.016444914042949677, "step": 1929 }, { "clip_ratio/high_max": 0.01119648793246597, "clip_ratio/high_mean": 0.01119648793246597, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.014767916523851454, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.19507121294736862, "epoch": 0.07751937984496124, "frac_reward_zero_std": 0.0, "grad_norm": 3.3098363876342773, "learning_rate": 4.154545454545455e-06, "loss": 0.0178, "num_tokens": 4355551.0, "reward": 0.84641432762146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.84641432762146, "reward_meter_std": 0.031771186739206314, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03177119046449661, "reward_total_composite_mean": 0.84641432762146, "reward_total_composite_std": 0.031771186739206314, "reward_total_mean": 0.84641432762146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.84641432762146, "rewards/meter/std": 0.031771186739206314, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.84641432762146, "rewards/total_composite/std": 0.031771186739206314, "sampling/importance_sampling_ratio/max": 1.4023113250732422, "sampling/importance_sampling_ratio/mean": 1.0048232078552246, "sampling/importance_sampling_ratio/min": 0.14267070591449738, "sampling/sampling_logp_difference/max": 1.9472160339355469, "sampling/sampling_logp_difference/mean": 0.027442539110779762, "step": 1930 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.008458438969682902, "epoch": 0.0775595453267462, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.151515151515152e-06, "loss": 0.0, "num_tokens": 4357311.0, "reward": 0.998939037322998, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998939037322998, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.998939037322998, "reward_total_composite_std": 0.0, "reward_total_mean": 0.998939037322998, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998939037322998, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998939037322998, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0244560241699219, "sampling/importance_sampling_ratio/mean": 1.0005451440811157, "sampling/importance_sampling_ratio/min": 0.9709616303443909, "sampling/sampling_logp_difference/max": 0.029468350112438202, "sampling/sampling_logp_difference/mean": 0.0009288773289881647, "step": 1931 }, { "clip_ratio/high_max": 0.024061617092229426, "clip_ratio/high_mean": 0.024061617092229426, "clip_ratio/low_mean": 0.0050675676902756095, "clip_ratio/low_min": 0.0050675676902756095, "clip_ratio/region_mean": 0.029129184782505035, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 73.125, "completions/mean_terminated_length": 73.125, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.17044041864573956, "epoch": 0.07759971080853115, "frac_reward_zero_std": 0.0, "grad_norm": 3.314406156539917, "learning_rate": 4.148484848484849e-06, "loss": 0.0098, "num_tokens": 4359128.0, "reward": 0.9976757764816284, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976757764816284, "reward_meter_std": 0.0007166875875554979, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007166775758378208, "reward_total_composite_mean": 0.9976757764816284, "reward_total_composite_std": 0.0007166875875554979, "reward_total_mean": 0.9976757764816284, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976757764816284, "rewards/meter/std": 0.0007166875875554979, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976757764816284, "rewards/total_composite/std": 0.0007166875875554979, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001599907875061, "sampling/importance_sampling_ratio/min": 0.26007840037345886, "sampling/sampling_logp_difference/max": 1.3467721939086914, "sampling/sampling_logp_difference/mean": 0.029741626232862473, "step": 1932 }, { "clip_ratio/high_max": 0.0155549660557881, "clip_ratio/high_mean": 0.0155549660557881, "clip_ratio/low_mean": 0.008469702675938606, "clip_ratio/low_min": 0.008469702675938606, "clip_ratio/region_mean": 0.024024668731726706, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 73.125, "completions/mean_terminated_length": 73.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.20059540774673223, "epoch": 0.0776398762903161, "frac_reward_zero_std": 0.0, "grad_norm": 4.072077751159668, "learning_rate": 4.145454545454546e-06, "loss": 0.0007, "num_tokens": 4361033.0, "reward": 0.9970039129257202, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970039129257202, "reward_meter_std": 0.0011512924684211612, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0011513038771227002, "reward_total_composite_mean": 0.9970039129257202, "reward_total_composite_std": 0.0011512924684211612, "reward_total_mean": 0.9970039129257202, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970039129257202, "rewards/meter/std": 0.0011512924684211612, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970039129257202, "rewards/total_composite/std": 0.0011512924684211612, "sampling/importance_sampling_ratio/max": 1.6174455881118774, "sampling/importance_sampling_ratio/mean": 1.0008363723754883, "sampling/importance_sampling_ratio/min": 0.2561119794845581, "sampling/sampling_logp_difference/max": 1.3621405363082886, "sampling/sampling_logp_difference/mean": 0.029263032600283623, "step": 1933 }, { "clip_ratio/high_max": 0.014982880558818579, "clip_ratio/high_mean": 0.014982880558818579, "clip_ratio/low_mean": 0.006601066328585148, "clip_ratio/low_min": 0.006601066328585148, "clip_ratio/region_mean": 0.021583946887403727, "completions/clipped_ratio": 0.0, "completions/max_length": 230.0, "completions/max_terminated_length": 230.0, "completions/mean_length": 226.0, "completions/mean_terminated_length": 226.0, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.10432031005620956, "epoch": 0.07768004177210105, "frac_reward_zero_std": 0.0, "grad_norm": 3.0962040424346924, "learning_rate": 4.142424242424243e-06, "loss": 0.0093, "num_tokens": 4364465.0, "reward": 0.6912178993225098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8571816086769104, "reward_meter_std": 0.171591117978096, "reward_repeat_penalty_mean": 0.8020833134651184, "reward_repeat_penalty_std": 0.0431290864944458, "reward_std": 0.1580277979373932, "reward_total_composite_mean": 0.6912178993225098, "reward_total_composite_std": 0.158027783036232, "reward_total_mean": 0.6912178993225098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8571816086769104, "rewards/meter/std": 0.171591117978096, "rewards/repeat_penalty/mean": 0.8020833134651184, "rewards/repeat_penalty/std": 0.0431290864944458, "rewards/total_composite/mean": 0.6912178993225098, "rewards/total_composite/std": 0.158027783036232, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9996902942657471, "sampling/importance_sampling_ratio/min": 0.0453910194337368, "sampling/sampling_logp_difference/max": 3.0924410820007324, "sampling/sampling_logp_difference/mean": 0.02388615906238556, "step": 1934 }, { "clip_ratio/high_max": 0.025740933255292475, "clip_ratio/high_mean": 0.025740933255292475, "clip_ratio/low_mean": 0.007359307492151856, "clip_ratio/low_min": 0.007359307492151856, "clip_ratio/region_mean": 0.03310024074744433, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.5, "completions/mean_terminated_length": 67.5, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.14524980820715427, "epoch": 0.07772020725388601, "frac_reward_zero_std": 0.0, "grad_norm": 3.562734365463257, "learning_rate": 4.13939393939394e-06, "loss": -0.0005, "num_tokens": 4366373.0, "reward": 0.72654128074646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.72654128074646, "reward_meter_std": 0.2970520853996277, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2970520555973053, "reward_total_composite_mean": 0.72654128074646, "reward_total_composite_std": 0.2970520853996277, "reward_total_mean": 0.72654128074646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.72654128074646, "rewards/meter/std": 0.2970520853996277, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.72654128074646, "rewards/total_composite/std": 0.2970520853996277, "sampling/importance_sampling_ratio/max": 1.5096417665481567, "sampling/importance_sampling_ratio/mean": 0.99456787109375, "sampling/importance_sampling_ratio/min": 0.24342434108257294, "sampling/sampling_logp_difference/max": 1.4129490852355957, "sampling/sampling_logp_difference/mean": 0.03413822129368782, "step": 1935 }, { "clip_ratio/high_max": 0.012465349864214659, "clip_ratio/high_mean": 0.012465349864214659, "clip_ratio/low_mean": 0.006351783173158765, "clip_ratio/low_min": 0.006351783173158765, "clip_ratio/region_mean": 0.018817133037373424, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 99.25, "completions/mean_terminated_length": 99.25, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.15317998873069882, "epoch": 0.07776037273567096, "frac_reward_zero_std": 0.0, "grad_norm": 3.389194965362549, "learning_rate": 4.136363636363637e-06, "loss": -0.0046, "num_tokens": 4368431.0, "reward": 0.7909245491027832, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8136383891105652, "reward_meter_std": 0.040068093687295914, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.028373470529913902, "reward_total_composite_mean": 0.7909245491027832, "reward_total_composite_std": 0.028373466804623604, "reward_total_mean": 0.7909245491027832, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8136383891105652, "rewards/meter/std": 0.040068093687295914, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7909245491027832, "rewards/total_composite/std": 0.028373466804623604, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0059889554977417, "sampling/importance_sampling_ratio/min": 0.4062268137931824, "sampling/sampling_logp_difference/max": 1.0089292526245117, "sampling/sampling_logp_difference/mean": 0.01888117380440235, "step": 1936 }, { "clip_ratio/high_max": 0.0021551724057644606, "clip_ratio/high_mean": 0.0021551724057644606, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0021551724057644606, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.018475122982636094, "epoch": 0.07780053821745592, "frac_reward_zero_std": 0.0, "grad_norm": 1.4040637016296387, "learning_rate": 4.133333333333333e-06, "loss": 0.0008, "num_tokens": 4370119.0, "reward": 0.994684100151062, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994684100151062, "reward_meter_std": 5.123758455738425e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.122838410898112e-05, "reward_total_composite_mean": 0.994684100151062, "reward_total_composite_std": 5.123758455738425e-05, "reward_total_mean": 0.994684100151062, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994684100151062, "rewards/meter/std": 5.123758455738425e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994684100151062, "rewards/total_composite/std": 5.123758455738425e-05, "sampling/importance_sampling_ratio/max": 1.146704077720642, "sampling/importance_sampling_ratio/mean": 0.9982123374938965, "sampling/importance_sampling_ratio/min": 0.6326279044151306, "sampling/sampling_logp_difference/max": 0.4578728675842285, "sampling/sampling_logp_difference/mean": 0.004311581142246723, "step": 1937 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.007407813798636198, "clip_ratio/low_min": 0.007407813798636198, "clip_ratio/region_mean": 0.007407813798636198, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.25, "completions/mean_terminated_length": 67.25, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.048515188973397017, "epoch": 0.07784070369924087, "frac_reward_zero_std": 0.0, "grad_norm": 3.30529522895813, "learning_rate": 4.1303030303030305e-06, "loss": -0.0008, "num_tokens": 4372001.0, "reward": 0.8598266839981079, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8598266839981079, "reward_meter_std": 0.02622251957654953, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02622251957654953, "reward_total_composite_mean": 0.8598266839981079, "reward_total_composite_std": 0.02622251957654953, "reward_total_mean": 0.8598266839981079, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8598266839981079, "rewards/meter/std": 0.02622251957654953, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8598266839981079, "rewards/total_composite/std": 0.02622251957654953, "sampling/importance_sampling_ratio/max": 1.5219361782073975, "sampling/importance_sampling_ratio/mean": 1.0010411739349365, "sampling/importance_sampling_ratio/min": 0.2974480092525482, "sampling/sampling_logp_difference/max": 1.2125158309936523, "sampling/sampling_logp_difference/mean": 0.011166814714670181, "step": 1938 }, { "clip_ratio/high_max": 0.02452767826616764, "clip_ratio/high_mean": 0.02452767826616764, "clip_ratio/low_mean": 0.0051369862630963326, "clip_ratio/low_min": 0.0051369862630963326, "clip_ratio/region_mean": 0.029664664529263973, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 72.0, "completions/mean_terminated_length": 72.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.28373553417623043, "epoch": 0.07788086918102582, "frac_reward_zero_std": 0.0, "grad_norm": 3.0253310203552246, "learning_rate": 4.127272727272728e-06, "loss": 0.0039, "num_tokens": 4373689.0, "reward": 0.8803695440292358, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8803695440292358, "reward_meter_std": 0.31208541989326477, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3120853900909424, "reward_total_composite_mean": 0.8803695440292358, "reward_total_composite_std": 0.31208541989326477, "reward_total_mean": 0.8803695440292358, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8803695440292358, "rewards/meter/std": 0.31208541989326477, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8803695440292358, "rewards/total_composite/std": 0.31208541989326477, "sampling/importance_sampling_ratio/max": 1.6593668460845947, "sampling/importance_sampling_ratio/mean": 1.0067641735076904, "sampling/importance_sampling_ratio/min": 0.3810611069202423, "sampling/sampling_logp_difference/max": 0.9647955894470215, "sampling/sampling_logp_difference/mean": 0.035320769995450974, "step": 1939 }, { "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018939394503831863, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.010286982113029808, "epoch": 0.07792103466281078, "frac_reward_zero_std": 0.0, "grad_norm": 0.002891062991693616, "learning_rate": 4.124242424242424e-06, "loss": -0.0001, "num_tokens": 4375497.0, "reward": 0.9989398717880249, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989398717880249, "reward_meter_std": 2.402423206149251e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.396394847892225e-06, "reward_total_composite_mean": 0.9989398717880249, "reward_total_composite_std": 2.402423206149251e-06, "reward_total_mean": 0.9989398717880249, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989398717880249, "rewards/meter/std": 2.402423206149251e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989398717880249, "rewards/total_composite/std": 2.402423206149251e-06, "sampling/importance_sampling_ratio/max": 1.0235236883163452, "sampling/importance_sampling_ratio/mean": 0.9999057054519653, "sampling/importance_sampling_ratio/min": 0.6075477004051208, "sampling/sampling_logp_difference/max": 0.4983246326446533, "sampling/sampling_logp_difference/mean": 0.0017837814521044493, "step": 1940 }, { "clip_ratio/high_max": 0.03567814873531461, "clip_ratio/high_mean": 0.03567814873531461, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.03567814873531461, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 489.0, "completions/mean_length": 472.75, "completions/mean_terminated_length": 467.14288330078125, "completions/min_length": 449.0, "completions/min_terminated_length": 449.0, "entropy": 0.5811280272901058, "epoch": 0.07796120014459573, "frac_reward_zero_std": 0.0, "grad_norm": 0.7915775775909424, "learning_rate": 4.1212121212121215e-06, "loss": -0.3041, "num_tokens": 4380383.0, "reward": 0.5711581707000732, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.6691176891326904, "reward_count_adherence_std": 0.062391772866249084, "reward_meter_mean": 0.9804998636245728, "reward_meter_std": 0.030505068600177765, "reward_repeat_penalty_mean": 0.9617094993591309, "reward_repeat_penalty_std": 0.04899247735738754, "reward_std": 0.2336382418870926, "reward_total_composite_mean": 0.5711581707000732, "reward_total_composite_std": 0.2336382418870926, "reward_total_mean": 0.5711581707000732, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.6691176891326904, "rewards/count_adherence/std": 0.062391772866249084, "rewards/meter/mean": 0.9804998636245728, "rewards/meter/std": 0.030505068600177765, "rewards/repeat_penalty/mean": 0.9617094993591309, "rewards/repeat_penalty/std": 0.04899247735738754, "rewards/total_composite/mean": 0.5711581707000732, "rewards/total_composite/std": 0.2336382418870926, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0137178897857666, "sampling/importance_sampling_ratio/min": 0.1862502098083496, "sampling/sampling_logp_difference/max": 1.680664300918579, "sampling/sampling_logp_difference/mean": 0.056584302335977554, "step": 1941 }, { "clip_ratio/high_max": 0.03670453419908881, "clip_ratio/high_mean": 0.03670453419908881, "clip_ratio/low_mean": 0.004934210330247879, "clip_ratio/low_min": 0.004934210330247879, "clip_ratio/region_mean": 0.04163874452933669, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 77.75, "completions/mean_terminated_length": 77.75, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.5831749886274338, "epoch": 0.07800136562638069, "frac_reward_zero_std": 0.0, "grad_norm": 4.029205799102783, "learning_rate": 4.118181818181819e-06, "loss": -0.008, "num_tokens": 4382253.0, "reward": 0.8649736642837524, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9882785081863403, "reward_meter_std": 0.018005145713686943, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.34996482729911804, "reward_total_composite_mean": 0.8649736642837524, "reward_total_composite_std": 0.34996482729911804, "reward_total_mean": 0.8649736642837524, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9882785081863403, "rewards/meter/std": 0.018005145713686943, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8649736642837524, "rewards/total_composite/std": 0.34996482729911804, "sampling/importance_sampling_ratio/max": 1.879008412361145, "sampling/importance_sampling_ratio/mean": 1.0108014345169067, "sampling/importance_sampling_ratio/min": 0.1804669201374054, "sampling/sampling_logp_difference/max": 1.7122077941894531, "sampling/sampling_logp_difference/mean": 0.06439422816038132, "step": 1942 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0012135922443121672, "clip_ratio/low_min": 0.0012135922443121672, "clip_ratio/region_mean": 0.0012135922443121672, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 96.0, "completions/mean_terminated_length": 96.0, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.030685836798511446, "epoch": 0.07804153110816564, "frac_reward_zero_std": 0.0, "grad_norm": 2.609006643295288, "learning_rate": 4.115151515151515e-06, "loss": 0.0281, "num_tokens": 4384317.0, "reward": 0.9721810817718506, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970648288726807, "reward_meter_std": 0.0006929152878001332, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07107476890087128, "reward_total_composite_mean": 0.9721810817718506, "reward_total_composite_std": 0.07107478380203247, "reward_total_mean": 0.9721810817718506, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970648288726807, "rewards/meter/std": 0.0006929152878001332, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9721810817718506, "rewards/total_composite/std": 0.07107478380203247, "sampling/importance_sampling_ratio/max": 1.3738577365875244, "sampling/importance_sampling_ratio/mean": 1.0022590160369873, "sampling/importance_sampling_ratio/min": 0.6829593181610107, "sampling/sampling_logp_difference/max": 0.3813199996948242, "sampling/sampling_logp_difference/mean": 0.00461819302290678, "step": 1943 }, { "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0019841270986944437, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.015167628531344235, "epoch": 0.0780816965899506, "frac_reward_zero_std": 0.0, "grad_norm": 0.053728338330984116, "learning_rate": 4.112121212121212e-06, "loss": -0.0001, "num_tokens": 4385917.0, "reward": 0.9970229864120483, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970229864120483, "reward_meter_std": 8.450447239738423e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.441417776339222e-06, "reward_total_composite_mean": 0.9970229864120483, "reward_total_composite_std": 8.450447239738423e-06, "reward_total_mean": 0.9970229864120483, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970229864120483, "rewards/meter/std": 8.450447239738423e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970229864120483, "rewards/total_composite/std": 8.450447239738423e-06, "sampling/importance_sampling_ratio/max": 1.081364393234253, "sampling/importance_sampling_ratio/mean": 1.0014358758926392, "sampling/importance_sampling_ratio/min": 0.8396610021591187, "sampling/sampling_logp_difference/max": 0.1747570037841797, "sampling/sampling_logp_difference/mean": 0.0021621037740260363, "step": 1944 }, { "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018939394503831863, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.019811521749943495, "epoch": 0.07812186207173555, "frac_reward_zero_std": 0.0, "grad_norm": 0.013028639368712902, "learning_rate": 4.10909090909091e-06, "loss": -0.0001, "num_tokens": 4387733.0, "reward": 0.9989398717880249, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989398717880249, "reward_meter_std": 2.402423206149251e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.396394847892225e-06, "reward_total_composite_mean": 0.9989398717880249, "reward_total_composite_std": 2.402423206149251e-06, "reward_total_mean": 0.9989398717880249, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989398717880249, "rewards/meter/std": 2.402423206149251e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989398717880249, "rewards/total_composite/std": 2.402423206149251e-06, "sampling/importance_sampling_ratio/max": 1.1326637268066406, "sampling/importance_sampling_ratio/mean": 1.0004053115844727, "sampling/importance_sampling_ratio/min": 0.8231624960899353, "sampling/sampling_logp_difference/max": 0.1946016550064087, "sampling/sampling_logp_difference/mean": 0.002988132182508707, "step": 1945 }, { "clip_ratio/high_max": 0.011114974273368716, "clip_ratio/high_mean": 0.011114974273368716, "clip_ratio/low_mean": 0.003623949596658349, "clip_ratio/low_min": 0.003623949596658349, "clip_ratio/region_mean": 0.014738923870027065, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.5, "completions/mean_terminated_length": 67.5, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.18513774313032627, "epoch": 0.0781620275535205, "frac_reward_zero_std": 0.0, "grad_norm": 2.780980110168457, "learning_rate": 4.106060606060606e-06, "loss": 0.0117, "num_tokens": 4389585.0, "reward": 0.8567509055137634, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8567509055137634, "reward_meter_std": 0.07118333131074905, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.07118332386016846, "reward_total_composite_mean": 0.8567509055137634, "reward_total_composite_std": 0.07118333131074905, "reward_total_mean": 0.8567509055137634, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8567509055137634, "rewards/meter/std": 0.07118333131074905, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8567509055137634, "rewards/total_composite/std": 0.07118333131074905, "sampling/importance_sampling_ratio/max": 1.5270895957946777, "sampling/importance_sampling_ratio/mean": 1.0084105730056763, "sampling/importance_sampling_ratio/min": 0.3204936981201172, "sampling/sampling_logp_difference/max": 1.137892723083496, "sampling/sampling_logp_difference/mean": 0.019841060042381287, "step": 1946 }, { "clip_ratio/high_max": 0.011963161756284535, "clip_ratio/high_mean": 0.011963161756284535, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.011963161756284535, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 73.125, "completions/mean_terminated_length": 73.125, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.1381580764427781, "epoch": 0.07820219303530546, "frac_reward_zero_std": 0.0, "grad_norm": 2.5665481090545654, "learning_rate": 4.103030303030303e-06, "loss": 0.003, "num_tokens": 4391466.0, "reward": 0.9979450702667236, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979450702667236, "reward_meter_std": 0.00038199007394723594, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00038197485264390707, "reward_total_composite_mean": 0.9979450702667236, "reward_total_composite_std": 0.00038199007394723594, "reward_total_mean": 0.9979450702667236, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979450702667236, "rewards/meter/std": 0.00038199007394723594, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979450702667236, "rewards/total_composite/std": 0.00038199007394723594, "sampling/importance_sampling_ratio/max": 1.5535597801208496, "sampling/importance_sampling_ratio/mean": 0.9995145201683044, "sampling/importance_sampling_ratio/min": 0.3299597203731537, "sampling/sampling_logp_difference/max": 1.1087846755981445, "sampling/sampling_logp_difference/mean": 0.02320614643394947, "step": 1947 }, { "clip_ratio/high_max": 0.012034508981741965, "clip_ratio/high_mean": 0.012034508981741965, "clip_ratio/low_mean": 0.00856354646384716, "clip_ratio/low_min": 0.00856354646384716, "clip_ratio/region_mean": 0.020598055445589125, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 73.375, "completions/mean_terminated_length": 73.375, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.16229457408189774, "epoch": 0.07824235851709041, "frac_reward_zero_std": 0.0, "grad_norm": 2.1724514961242676, "learning_rate": 4.1e-06, "loss": 0.0029, "num_tokens": 4393381.0, "reward": 0.9980074167251587, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980074167251587, "reward_meter_std": 0.0006349986069835722, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006350158364512026, "reward_total_composite_mean": 0.9980074167251587, "reward_total_composite_std": 0.0006349986069835722, "reward_total_mean": 0.9980074167251587, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980074167251587, "rewards/meter/std": 0.0006349986069835722, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980074167251587, "rewards/total_composite/std": 0.0006349986069835722, "sampling/importance_sampling_ratio/max": 1.8275814056396484, "sampling/importance_sampling_ratio/mean": 1.0009372234344482, "sampling/importance_sampling_ratio/min": 0.4539879858493805, "sampling/sampling_logp_difference/max": 0.789684534072876, "sampling/sampling_logp_difference/mean": 0.025800568982958794, "step": 1948 }, { "clip_ratio/high_max": 0.012344544753432274, "clip_ratio/high_mean": 0.012344544753432274, "clip_ratio/low_mean": 0.023918771417811513, "clip_ratio/low_min": 0.023918771417811513, "clip_ratio/region_mean": 0.03626331617124379, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 122.5, "completions/mean_terminated_length": 122.5, "completions/min_length": 110.0, "completions/min_terminated_length": 110.0, "entropy": 0.1550168227404356, "epoch": 0.07828252399887536, "frac_reward_zero_std": 0.0, "grad_norm": 5.015791893005371, "learning_rate": 4.096969696969697e-06, "loss": -0.0289, "num_tokens": 4395881.0, "reward": 0.0730920359492302, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1157275140285492, "reward_meter_mean": 0.10345004498958588, "reward_meter_std": 0.17283838987350464, "reward_repeat_penalty_mean": 0.851190447807312, "reward_repeat_penalty_std": 0.08266931027173996, "reward_std": 0.11985667049884796, "reward_total_composite_mean": 0.0730920359492302, "reward_total_composite_std": 0.11985667049884796, "reward_total_mean": 0.0730920359492302, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1157275140285492, "rewards/meter/mean": 0.10345004498958588, "rewards/meter/std": 0.17283838987350464, "rewards/repeat_penalty/mean": 0.851190447807312, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.0730920359492302, "rewards/total_composite/std": 0.11985667049884796, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9997954368591309, "sampling/importance_sampling_ratio/min": 0.2712607979774475, "sampling/sampling_logp_difference/max": 1.3046746253967285, "sampling/sampling_logp_difference/mean": 0.03781670331954956, "step": 1949 }, { "clip_ratio/high_max": 0.0033783784601837397, "clip_ratio/high_mean": 0.0033783784601837397, "clip_ratio/low_mean": 0.00868320802692324, "clip_ratio/low_min": 0.00868320802692324, "clip_ratio/region_mean": 0.012061586487106979, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 54.875, "completions/mean_terminated_length": 54.875, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.1553973164409399, "epoch": 0.07832268948066032, "frac_reward_zero_std": 0.0, "grad_norm": 6.135134220123291, "learning_rate": 4.093939393939394e-06, "loss": 0.3165, "num_tokens": 4397640.0, "reward": 0.49901658296585083, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5, "reward_count_adherence_std": 0.5345224738121033, "reward_meter_mean": 0.9981762170791626, "reward_meter_std": 0.0004922283114865422, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.5334712862968445, "reward_total_composite_mean": 0.49901658296585083, "reward_total_composite_std": 0.5334713459014893, "reward_total_mean": 0.49901658296585083, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5, "rewards/count_adherence/std": 0.5345224738121033, "rewards/meter/mean": 0.9981762170791626, "rewards/meter/std": 0.0004922283114865422, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.49901658296585083, "rewards/total_composite/std": 0.5334713459014893, "sampling/importance_sampling_ratio/max": 1.6159636974334717, "sampling/importance_sampling_ratio/mean": 1.0079329013824463, "sampling/importance_sampling_ratio/min": 0.22332191467285156, "sampling/sampling_logp_difference/max": 1.499140977859497, "sampling/sampling_logp_difference/mean": 0.02422521449625492, "step": 1950 }, { "epoch": 0.07832268948066032, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/max_length": 383.15384615384613, "eval_completions/max_terminated_length": 372.84615384615387, "eval_completions/mean_length": 219.20192307692307, "eval_completions/mean_terminated_length": 216.13324209359976, "eval_completions/min_length": 67.84615384615384, "eval_completions/min_terminated_length": 67.84615384615384, "eval_entropy": 0.2521360000738731, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 4397640.0, "eval_reward": 0.5268221841408656, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.8801831007003784, "eval_reward_count_adherence_std": 0.15413257680260217, "eval_reward_meter_mean": 0.7193602369381831, "eval_reward_meter_std": 0.41060519218444824, "eval_reward_repeat_penalty_mean": 0.8391693326143118, "eval_reward_repeat_penalty_std": 0.14583237583820635, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5268221841408656, "eval_reward_total_composite_std": 0.36114954260679394, "eval_reward_total_mean": 0.5268221841408656, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.8801831007003784, "eval_rewards/count_adherence/std": 0.15413257680260217, "eval_rewards/meter/mean": 0.7193602369381831, "eval_rewards/meter/std": 0.41060519218444824, "eval_rewards/repeat_penalty/mean": 0.8391693326143118, "eval_rewards/repeat_penalty/std": 0.14583237583820635, "eval_rewards/total_composite/mean": 0.5268221841408656, "eval_rewards/total_composite/std": 0.36114954260679394, "eval_runtime": 74.1396, "eval_samples_per_second": 1.403, "eval_sampling/importance_sampling_ratio/max": 1.5754675223277166, "eval_sampling/importance_sampling_ratio/mean": 1.0057256405170147, "eval_sampling/importance_sampling_ratio/min": 0.3655441105365753, "eval_sampling/sampling_logp_difference/max": 1.0237883787888746, "eval_sampling/sampling_logp_difference/mean": 0.02039779959103236, "eval_steps_per_second": 0.175, "step": 1950 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0027777778450399637, "clip_ratio/low_min": 0.0027777778450399637, "clip_ratio/region_mean": 0.0027777778450399637, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 90.0, "completions/mean_terminated_length": 90.0, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.015661518438719213, "epoch": 0.07836285496244527, "frac_reward_zero_std": 0.0, "grad_norm": 0.36703649163246155, "learning_rate": 4.0909090909090915e-06, "loss": 0.0003, "num_tokens": 4399744.0, "reward": 0.9955518841743469, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955518841743469, "reward_meter_std": 1.7999940610025078e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.8004628145718016e-05, "reward_total_composite_mean": 0.9955518841743469, "reward_total_composite_std": 1.7999940610025078e-05, "reward_total_mean": 0.9955518841743469, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955518841743469, "rewards/meter/std": 1.7999940610025078e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955518841743469, "rewards/total_composite/std": 1.7999940610025078e-05, "sampling/importance_sampling_ratio/max": 1.191457748413086, "sampling/importance_sampling_ratio/mean": 0.9992460012435913, "sampling/importance_sampling_ratio/min": 0.40764331817626953, "sampling/sampling_logp_difference/max": 0.8973627090454102, "sampling/sampling_logp_difference/mean": 0.004722072742879391, "step": 1951 }, { "clip_ratio/high_max": 0.031067086150869727, "clip_ratio/high_mean": 0.031067086150869727, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.031067086150869727, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.875, "completions/mean_terminated_length": 71.875, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.23499170877039433, "epoch": 0.07840302044423023, "frac_reward_zero_std": 0.0, "grad_norm": 4.403555870056152, "learning_rate": 4.087878787878789e-06, "loss": -0.0014, "num_tokens": 4401615.0, "reward": 0.987866997718811, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.987866997718811, "reward_meter_std": 0.015270087867975235, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.015270096249878407, "reward_total_composite_mean": 0.987866997718811, "reward_total_composite_std": 0.015270087867975235, "reward_total_mean": 0.987866997718811, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.987866997718811, "rewards/meter/std": 0.015270087867975235, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.987866997718811, "rewards/total_composite/std": 0.015270087867975235, "sampling/importance_sampling_ratio/max": 1.4481229782104492, "sampling/importance_sampling_ratio/mean": 1.002774715423584, "sampling/importance_sampling_ratio/min": 0.43718650937080383, "sampling/sampling_logp_difference/max": 0.8273954391479492, "sampling/sampling_logp_difference/mean": 0.028592025861144066, "step": 1952 }, { "clip_ratio/high_max": 0.029118790524080396, "clip_ratio/high_mean": 0.029118790524080396, "clip_ratio/low_mean": 0.012907008291222155, "clip_ratio/low_min": 0.012907008291222155, "clip_ratio/region_mean": 0.04202579881530255, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 74.5, "completions/mean_terminated_length": 74.5, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.2592748161405325, "epoch": 0.07844318592601518, "frac_reward_zero_std": 0.0, "grad_norm": 6.583865165710449, "learning_rate": 4.084848484848485e-06, "loss": 0.0586, "num_tokens": 4403483.0, "reward": 0.41915929317474365, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.43240049481391907, "reward_meter_std": 0.30809712409973145, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.30558669567108154, "reward_total_composite_mean": 0.41915929317474365, "reward_total_composite_std": 0.30558672547340393, "reward_total_mean": 0.41915929317474365, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.43240049481391907, "rewards/meter/std": 0.30809712409973145, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.41915929317474365, "rewards/total_composite/std": 0.30558672547340393, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0041871070861816, "sampling/importance_sampling_ratio/min": 0.3223821520805359, "sampling/sampling_logp_difference/max": 1.1320176124572754, "sampling/sampling_logp_difference/mean": 0.043784063309431076, "step": 1953 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.02731157885864377, "epoch": 0.07848335140780013, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.081818181818182e-06, "loss": 0.0, "num_tokens": 4405203.0, "reward": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989815950393677, "reward_meter_std": 5.247284934739582e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.0, "reward_total_composite_std": 0.0, "reward_total_mean": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989815950393677, "rewards/meter/std": 5.247284934739582e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.0, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.501068353652954, "sampling/importance_sampling_ratio/mean": 1.001094937324524, "sampling/importance_sampling_ratio/min": 0.6015753746032715, "sampling/sampling_logp_difference/max": 0.5082035064697266, "sampling/sampling_logp_difference/mean": 0.007122131530195475, "step": 1954 }, { "clip_ratio/high_max": 0.005274800234474242, "clip_ratio/high_mean": 0.005274800234474242, "clip_ratio/low_mean": 0.007172186364186928, "clip_ratio/low_min": 0.007172186364186928, "clip_ratio/region_mean": 0.01244698659866117, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 300.375, "completions/mean_terminated_length": 300.375, "completions/min_length": 293.0, "completions/min_terminated_length": 293.0, "entropy": 0.12408511992543936, "epoch": 0.07852351688958509, "frac_reward_zero_std": 0.0, "grad_norm": 1.9077023267745972, "learning_rate": 4.07878787878788e-06, "loss": -0.0135, "num_tokens": 4409222.0, "reward": 0.6738342642784119, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946515560150146, "reward_meter_std": 0.002225227653980255, "reward_repeat_penalty_mean": 0.7741013169288635, "reward_repeat_penalty_std": 0.08203362673521042, "reward_std": 0.07273106276988983, "reward_total_composite_mean": 0.6738342642784119, "reward_total_composite_std": 0.07273107022047043, "reward_total_mean": 0.6738342642784119, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946515560150146, "rewards/meter/std": 0.002225227653980255, "rewards/repeat_penalty/mean": 0.7741013169288635, "rewards/repeat_penalty/std": 0.08203362673521042, "rewards/total_composite/mean": 0.6738342642784119, "rewards/total_composite/std": 0.07273107022047043, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0040055513381958, "sampling/importance_sampling_ratio/min": 0.08110546320676804, "sampling/sampling_logp_difference/max": 2.512004852294922, "sampling/sampling_logp_difference/mean": 0.018673671409487724, "step": 1955 }, { "clip_ratio/high_max": 0.041018035262823105, "clip_ratio/high_mean": 0.041018035262823105, "clip_ratio/low_mean": 0.028407905949279666, "clip_ratio/low_min": 0.028407905949279666, "clip_ratio/region_mean": 0.06942594121210277, "completions/clipped_ratio": 0.0, "completions/max_length": 51.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 47.375, "completions/mean_terminated_length": 47.375, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.3848374430090189, "epoch": 0.07856368237137004, "frac_reward_zero_std": 0.0, "grad_norm": 8.814888000488281, "learning_rate": 4.075757575757576e-06, "loss": 0.0496, "num_tokens": 4410857.0, "reward": 0.6137043237686157, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6137043237686157, "reward_meter_std": 0.33018791675567627, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.33018791675567627, "reward_total_composite_mean": 0.6137043237686157, "reward_total_composite_std": 0.33018791675567627, "reward_total_mean": 0.6137043237686157, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6137043237686157, "rewards/meter/std": 0.33018791675567627, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6137043237686157, "rewards/total_composite/std": 0.33018791675567627, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9924876689910889, "sampling/importance_sampling_ratio/min": 0.18539191782474518, "sampling/sampling_logp_difference/max": 1.6852831840515137, "sampling/sampling_logp_difference/mean": 0.06914983689785004, "step": 1956 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.009184467780869454, "clip_ratio/low_min": 0.009184467780869454, "clip_ratio/region_mean": 0.011022703081835061, "completions/clipped_ratio": 0.0, "completions/max_length": 139.0, "completions/max_terminated_length": 139.0, "completions/mean_length": 136.75, "completions/mean_terminated_length": 136.75, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "entropy": 0.0967075265944004, "epoch": 0.078603847853155, "frac_reward_zero_std": 0.0, "grad_norm": 2.345047950744629, "learning_rate": 4.072727272727273e-06, "loss": 0.0042, "num_tokens": 4413295.0, "reward": 0.8717420101165771, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962515830993652, "reward_meter_std": 0.0024396260268986225, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05084508657455444, "reward_total_composite_mean": 0.8717420101165771, "reward_total_composite_std": 0.05084509775042534, "reward_total_mean": 0.8717420101165771, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962515830993652, "rewards/meter/std": 0.0024396260268986225, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8717420101165771, "rewards/total_composite/std": 0.05084509775042534, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0027059316635132, "sampling/importance_sampling_ratio/min": 0.045186638832092285, "sampling/sampling_logp_difference/max": 3.096953868865967, "sampling/sampling_logp_difference/mean": 0.01802092418074608, "step": 1957 }, { "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007575757801532745, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 33.0, "completions/mean_terminated_length": 33.0, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.20033110305666924, "epoch": 0.07864401333493995, "frac_reward_zero_std": 0.0, "grad_norm": 10.1416654586792, "learning_rate": 4.0696969696969706e-06, "loss": 0.0012, "num_tokens": 4414671.0, "reward": 0.9584056735038757, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9584056735038757, "reward_meter_std": 0.020386451855301857, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.020386451855301857, "reward_total_composite_mean": 0.9584056735038757, "reward_total_composite_std": 0.020386451855301857, "reward_total_mean": 0.9584056735038757, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9584056735038757, "rewards/meter/std": 0.020386451855301857, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9584056735038757, "rewards/total_composite/std": 0.020386451855301857, "sampling/importance_sampling_ratio/max": 1.6090956926345825, "sampling/importance_sampling_ratio/mean": 1.0064359903335571, "sampling/importance_sampling_ratio/min": 0.37418755888938904, "sampling/sampling_logp_difference/max": 0.9829981327056885, "sampling/sampling_logp_difference/mean": 0.025948278605937958, "step": 1958 }, { "clip_ratio/high_max": 0.028867413057014346, "clip_ratio/high_mean": 0.028867413057014346, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.03243884164839983, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 76.25, "completions/mean_terminated_length": 76.25, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.44979994744062424, "epoch": 0.0786841788167249, "frac_reward_zero_std": 0.0, "grad_norm": 8.695691108703613, "learning_rate": 4.066666666666667e-06, "loss": -0.0107, "num_tokens": 4416753.0, "reward": 0.9713230133056641, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9713230133056641, "reward_meter_std": 0.06465231627225876, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06465231627225876, "reward_total_composite_mean": 0.9713230133056641, "reward_total_composite_std": 0.06465231627225876, "reward_total_mean": 0.9713230133056641, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9713230133056641, "rewards/meter/std": 0.06465231627225876, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9713230133056641, "rewards/total_composite/std": 0.06465231627225876, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0147606134414673, "sampling/importance_sampling_ratio/min": 0.303402304649353, "sampling/sampling_logp_difference/max": 1.1926956176757812, "sampling/sampling_logp_difference/mean": 0.04479827731847763, "step": 1959 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.007722355891019106, "clip_ratio/low_min": 0.007722355891019106, "clip_ratio/region_mean": 0.01145369908772409, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.253327926620841, "epoch": 0.07872434429850986, "frac_reward_zero_std": 0.0, "grad_norm": 5.968605041503906, "learning_rate": 4.063636363636364e-06, "loss": -0.0106, "num_tokens": 4418613.0, "reward": 0.7400355339050293, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7400355339050293, "reward_meter_std": 0.26013627648353577, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.26013627648353577, "reward_total_composite_mean": 0.7400355339050293, "reward_total_composite_std": 0.26013627648353577, "reward_total_mean": 0.7400355339050293, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7400355339050293, "rewards/meter/std": 0.26013627648353577, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7400355339050293, "rewards/total_composite/std": 0.26013627648353577, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0059431791305542, "sampling/importance_sampling_ratio/min": 0.1926136314868927, "sampling/sampling_logp_difference/max": 1.647068977355957, "sampling/sampling_logp_difference/mean": 0.03610033914446831, "step": 1960 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.014339913497678936, "epoch": 0.07876450978029481, "frac_reward_zero_std": 0.0, "grad_norm": 0.6822566390037537, "learning_rate": 4.060606060606061e-06, "loss": -0.0019, "num_tokens": 4420309.0, "reward": 0.996955156326294, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996955156326294, "reward_meter_std": 0.0001835073926486075, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00018350737809669226, "reward_total_composite_mean": 0.996955156326294, "reward_total_composite_std": 0.0001835073926486075, "reward_total_mean": 0.996955156326294, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996955156326294, "rewards/meter/std": 0.0001835073926486075, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996955156326294, "rewards/total_composite/std": 0.0001835073926486075, "sampling/importance_sampling_ratio/max": 1.0358319282531738, "sampling/importance_sampling_ratio/mean": 1.0001448392868042, "sampling/importance_sampling_ratio/min": 0.36827507615089417, "sampling/sampling_logp_difference/max": 0.9989252090454102, "sampling/sampling_logp_difference/mean": 0.003504997817799449, "step": 1961 }, { "clip_ratio/high_max": 0.01814535935409367, "clip_ratio/high_mean": 0.01814535935409367, "clip_ratio/low_mean": 0.006172839552164078, "clip_ratio/low_min": 0.006172839552164078, "clip_ratio/region_mean": 0.02431819890625775, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 77.5, "completions/mean_terminated_length": 77.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.5117241851985455, "epoch": 0.07880467526207977, "frac_reward_zero_std": 0.0, "grad_norm": 3.0988998413085938, "learning_rate": 4.057575757575758e-06, "loss": 0.0256, "num_tokens": 4422201.0, "reward": 0.9931849837303162, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9931849837303162, "reward_meter_std": 0.005035399924963713, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00503541948273778, "reward_total_composite_mean": 0.9931849837303162, "reward_total_composite_std": 0.005035399924963713, "reward_total_mean": 0.9931849837303162, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9931849837303162, "rewards/meter/std": 0.005035399924963713, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931849837303162, "rewards/total_composite/std": 0.005035399924963713, "sampling/importance_sampling_ratio/max": 1.963693380355835, "sampling/importance_sampling_ratio/mean": 1.0084387063980103, "sampling/importance_sampling_ratio/min": 0.30108973383903503, "sampling/sampling_logp_difference/max": 1.2003469467163086, "sampling/sampling_logp_difference/mean": 0.04507505148649216, "step": 1962 }, { "clip_ratio/high_max": 0.018698629923164845, "clip_ratio/high_mean": 0.018698629923164845, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/region_mean": 0.022170852171257138, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 72.625, "completions/mean_terminated_length": 72.625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.245115103200078, "epoch": 0.07884484074386472, "frac_reward_zero_std": 0.0, "grad_norm": 5.142487049102783, "learning_rate": 4.054545454545455e-06, "loss": -0.01, "num_tokens": 4423998.0, "reward": 0.997757077217102, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997757077217102, "reward_meter_std": 0.0013475378509610891, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0013475335435941815, "reward_total_composite_mean": 0.997757077217102, "reward_total_composite_std": 0.0013475378509610891, "reward_total_mean": 0.997757077217102, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997757077217102, "rewards/meter/std": 0.0013475378509610891, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997757077217102, "rewards/total_composite/std": 0.0013475378509610891, "sampling/importance_sampling_ratio/max": 1.6228265762329102, "sampling/importance_sampling_ratio/mean": 0.9945013523101807, "sampling/importance_sampling_ratio/min": 0.3417564034461975, "sampling/sampling_logp_difference/max": 1.0736570358276367, "sampling/sampling_logp_difference/mean": 0.0390477180480957, "step": 1963 }, { "clip_ratio/high_max": 0.009661172516644001, "clip_ratio/high_mean": 0.009661172516644001, "clip_ratio/low_mean": 0.001377504551783204, "clip_ratio/low_min": 0.001377504551783204, "clip_ratio/region_mean": 0.011038677068427205, "completions/clipped_ratio": 0.0, "completions/max_length": 183.0, "completions/max_terminated_length": 183.0, "completions/mean_length": 181.0, "completions/mean_terminated_length": 181.0, "completions/min_length": 180.0, "completions/min_terminated_length": 180.0, "entropy": 0.1502722892910242, "epoch": 0.07888500622564967, "frac_reward_zero_std": 0.0, "grad_norm": 2.332364320755005, "learning_rate": 4.0515151515151516e-06, "loss": 0.0047, "num_tokens": 4426910.0, "reward": 0.8455538749694824, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980403780937195, "reward_meter_std": 0.00041446235263720155, "reward_repeat_penalty_mean": 0.8472222685813904, "reward_repeat_penalty_std": 0.08266931027173996, "reward_std": 0.08242160081863403, "reward_total_composite_mean": 0.8455538749694824, "reward_total_composite_std": 0.08242159336805344, "reward_total_mean": 0.8455538749694824, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980403780937195, "rewards/meter/std": 0.00041446235263720155, "rewards/repeat_penalty/mean": 0.8472222685813904, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.8455538749694824, "rewards/total_composite/std": 0.08242159336805344, "sampling/importance_sampling_ratio/max": 1.5945336818695068, "sampling/importance_sampling_ratio/mean": 1.0034029483795166, "sampling/importance_sampling_ratio/min": 0.17582646012306213, "sampling/sampling_logp_difference/max": 1.738257884979248, "sampling/sampling_logp_difference/mean": 0.021872445940971375, "step": 1964 }, { "clip_ratio/high_max": 0.024971136124804616, "clip_ratio/high_mean": 0.024971136124804616, "clip_ratio/low_mean": 0.007042253389954567, "clip_ratio/low_min": 0.007042253389954567, "clip_ratio/region_mean": 0.03201338951475918, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 74.5, "completions/mean_terminated_length": 74.5, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.43255916610360146, "epoch": 0.07892517170743463, "frac_reward_zero_std": 0.0, "grad_norm": 7.793933391571045, "learning_rate": 4.048484848484849e-06, "loss": -0.0106, "num_tokens": 4428746.0, "reward": 0.9973430037498474, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973430037498474, "reward_meter_std": 0.001286112586967647, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0012860975693911314, "reward_total_composite_mean": 0.9973430037498474, "reward_total_composite_std": 0.001286112586967647, "reward_total_mean": 0.9973430037498474, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973430037498474, "rewards/meter/std": 0.001286112586967647, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973430037498474, "rewards/total_composite/std": 0.001286112586967647, "sampling/importance_sampling_ratio/max": 1.865578293800354, "sampling/importance_sampling_ratio/mean": 1.0060445070266724, "sampling/importance_sampling_ratio/min": 0.11871980875730515, "sampling/sampling_logp_difference/max": 2.1309890747070312, "sampling/sampling_logp_difference/mean": 0.05123698338866234, "step": 1965 }, { "clip_ratio/high_max": 0.027550982777029276, "clip_ratio/high_mean": 0.027550982777029276, "clip_ratio/low_mean": 0.00909090880304575, "clip_ratio/low_min": 0.00909090880304575, "clip_ratio/region_mean": 0.036641891580075026, "completions/clipped_ratio": 0.0, "completions/max_length": 112.0, "completions/max_terminated_length": 112.0, "completions/mean_length": 109.25, "completions/mean_terminated_length": 109.25, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.38045662455260754, "epoch": 0.07896533718921958, "frac_reward_zero_std": 0.0, "grad_norm": 2.1039681434631348, "learning_rate": 4.045454545454546e-06, "loss": 0.0066, "num_tokens": 4430956.0, "reward": 0.9682018160820007, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.993117094039917, "reward_meter_std": 0.010319688357412815, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06981158256530762, "reward_total_composite_mean": 0.9682018160820007, "reward_total_composite_std": 0.06981157511472702, "reward_total_mean": 0.9682018160820007, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.993117094039917, "rewards/meter/std": 0.010319688357412815, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9682018160820007, "rewards/total_composite/std": 0.06981157511472702, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0064747333526611, "sampling/importance_sampling_ratio/min": 0.2549518346786499, "sampling/sampling_logp_difference/max": 1.36668062210083, "sampling/sampling_logp_difference/mean": 0.04413502663373947, "step": 1966 }, { "clip_ratio/high_max": 0.019368293695151806, "clip_ratio/high_mean": 0.019368293695151806, "clip_ratio/low_mean": 0.022320898715406656, "clip_ratio/low_min": 0.022320898715406656, "clip_ratio/region_mean": 0.04168919241055846, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 71.125, "completions/mean_terminated_length": 71.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2595768254250288, "epoch": 0.07900550267100453, "frac_reward_zero_std": 0.0, "grad_norm": 2.3863680362701416, "learning_rate": 4.0424242424242425e-06, "loss": 0.0108, "num_tokens": 4432773.0, "reward": 0.9883999824523926, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9883999824523926, "reward_meter_std": 0.005968651734292507, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005968641955405474, "reward_total_composite_mean": 0.9883999824523926, "reward_total_composite_std": 0.005968651734292507, "reward_total_mean": 0.9883999824523926, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9883999824523926, "rewards/meter/std": 0.005968651734292507, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9883999824523926, "rewards/total_composite/std": 0.005968651734292507, "sampling/importance_sampling_ratio/max": 1.7200582027435303, "sampling/importance_sampling_ratio/mean": 1.0037516355514526, "sampling/importance_sampling_ratio/min": 0.24122373759746552, "sampling/sampling_logp_difference/max": 1.4220304489135742, "sampling/sampling_logp_difference/mean": 0.04225626587867737, "step": 1967 }, { "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.011363636702299118, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.125, "completions/mean_terminated_length": 33.125, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.09357678238302469, "epoch": 0.07904566815278949, "frac_reward_zero_std": 0.0, "grad_norm": 3.711296319961548, "learning_rate": 4.03939393939394e-06, "loss": 0.0097, "num_tokens": 4434286.0, "reward": 0.9655854105949402, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9655854105949402, "reward_meter_std": 0.0028627016581594944, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0028626781422644854, "reward_total_composite_mean": 0.9655854105949402, "reward_total_composite_std": 0.0028627016581594944, "reward_total_mean": 0.9655854105949402, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9655854105949402, "rewards/meter/std": 0.0028627016581594944, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9655854105949402, "rewards/total_composite/std": 0.0028627016581594944, "sampling/importance_sampling_ratio/max": 1.5242719650268555, "sampling/importance_sampling_ratio/mean": 1.0038316249847412, "sampling/importance_sampling_ratio/min": 0.5865768194198608, "sampling/sampling_logp_difference/max": 0.5334515571594238, "sampling/sampling_logp_difference/mean": 0.012964806519448757, "step": 1968 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.012546915793791413, "epoch": 0.07908583363457444, "frac_reward_zero_std": 0.0, "grad_norm": 0.08461344242095947, "learning_rate": 4.036363636363637e-06, "loss": 0.0, "num_tokens": 4436078.0, "reward": 0.9970186352729797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970186352729797, "reward_meter_std": 3.89859178540064e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.889565959980246e-06, "reward_total_composite_mean": 0.9970186352729797, "reward_total_composite_std": 3.89859178540064e-06, "reward_total_mean": 0.9970186352729797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970186352729797, "rewards/meter/std": 3.89859178540064e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970186352729797, "rewards/total_composite/std": 3.89859178540064e-06, "sampling/importance_sampling_ratio/max": 1.0470123291015625, "sampling/importance_sampling_ratio/mean": 1.0005252361297607, "sampling/importance_sampling_ratio/min": 0.5888224244117737, "sampling/sampling_logp_difference/max": 0.5296306610107422, "sampling/sampling_logp_difference/mean": 0.002396326744928956, "step": 1969 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/region_mean": 0.011363636702299118, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 33.0, "completions/mean_terminated_length": 33.0, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.08069896046072245, "epoch": 0.0791259991163594, "frac_reward_zero_std": 0.0, "grad_norm": 4.984862327575684, "learning_rate": 4.033333333333333e-06, "loss": 0.0007, "num_tokens": 4437774.0, "reward": 0.9650729894638062, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9650729894638062, "reward_meter_std": 0.0013392050750553608, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0013392041437327862, "reward_total_composite_mean": 0.9650729894638062, "reward_total_composite_std": 0.0013392050750553608, "reward_total_mean": 0.9650729894638062, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9650729894638062, "rewards/meter/std": 0.0013392050750553608, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9650729894638062, "rewards/total_composite/std": 0.0013392050750553608, "sampling/importance_sampling_ratio/max": 1.5502804517745972, "sampling/importance_sampling_ratio/mean": 1.0015711784362793, "sampling/importance_sampling_ratio/min": 0.5409092307090759, "sampling/sampling_logp_difference/max": 0.6145038604736328, "sampling/sampling_logp_difference/mean": 0.010843876749277115, "step": 1970 }, { "clip_ratio/high_max": 0.01056614622939378, "clip_ratio/high_mean": 0.01056614622939378, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/region_mean": 0.019127790001221, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.25, "completions/mean_terminated_length": 72.25, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.24298556335270405, "epoch": 0.07916616459814435, "frac_reward_zero_std": 0.0, "grad_norm": 2.97381854057312, "learning_rate": 4.030303030303031e-06, "loss": 0.0063, "num_tokens": 4439592.0, "reward": 0.9976744651794434, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976744651794434, "reward_meter_std": 0.0007276128162629902, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007276106043718755, "reward_total_composite_mean": 0.9976744651794434, "reward_total_composite_std": 0.0007276128162629902, "reward_total_mean": 0.9976744651794434, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976744651794434, "rewards/meter/std": 0.0007276128162629902, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976744651794434, "rewards/total_composite/std": 0.0007276128162629902, "sampling/importance_sampling_ratio/max": 1.839429497718811, "sampling/importance_sampling_ratio/mean": 1.0109328031539917, "sampling/importance_sampling_ratio/min": 0.2639040946960449, "sampling/sampling_logp_difference/max": 1.332169532775879, "sampling/sampling_logp_difference/mean": 0.03132731840014458, "step": 1971 }, { "clip_ratio/high_max": 0.02989264251664281, "clip_ratio/high_mean": 0.02989264251664281, "clip_ratio/low_mean": 0.015499898232519627, "clip_ratio/low_min": 0.015499898232519627, "clip_ratio/region_mean": 0.045392540749162436, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 75.125, "completions/mean_terminated_length": 75.125, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.47906696051359177, "epoch": 0.0792063300799293, "frac_reward_zero_std": 0.0, "grad_norm": 5.1056904792785645, "learning_rate": 4.027272727272727e-06, "loss": -0.0272, "num_tokens": 4441521.0, "reward": 0.9973230361938477, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973230361938477, "reward_meter_std": 0.0019267015159130096, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0019267020979896188, "reward_total_composite_mean": 0.9973230361938477, "reward_total_composite_std": 0.0019267015159130096, "reward_total_mean": 0.9973230361938477, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973230361938477, "rewards/meter/std": 0.0019267015159130096, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973230361938477, "rewards/total_composite/std": 0.0019267015159130096, "sampling/importance_sampling_ratio/max": 1.9101423025131226, "sampling/importance_sampling_ratio/mean": 1.0059734582901, "sampling/importance_sampling_ratio/min": 0.25476324558258057, "sampling/sampling_logp_difference/max": 1.3674206733703613, "sampling/sampling_logp_difference/mean": 0.04758862033486366, "step": 1972 }, { "clip_ratio/high_max": 0.003703703638166189, "clip_ratio/high_mean": 0.003703703638166189, "clip_ratio/low_mean": 0.0028056252049282193, "clip_ratio/low_min": 0.0028056252049282193, "clip_ratio/region_mean": 0.0065093288430944085, "completions/clipped_ratio": 0.0, "completions/max_length": 135.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 134.75, "completions/mean_terminated_length": 134.75, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.0543739995919168, "epoch": 0.07924649556171426, "frac_reward_zero_std": 0.0, "grad_norm": 1.4667316675186157, "learning_rate": 4.024242424242424e-06, "loss": -0.0034, "num_tokens": 4443935.0, "reward": 0.7524703145027161, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8778820037841797, "reward_meter_std": 0.01677864044904709, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014381689950823784, "reward_total_composite_mean": 0.7524703145027161, "reward_total_composite_std": 0.014381703920662403, "reward_total_mean": 0.7524703145027161, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8778820037841797, "rewards/meter/std": 0.01677864044904709, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7524703145027161, "rewards/total_composite/std": 0.014381703920662403, "sampling/importance_sampling_ratio/max": 1.7678629159927368, "sampling/importance_sampling_ratio/mean": 1.0035121440887451, "sampling/importance_sampling_ratio/min": 0.21779848635196686, "sampling/sampling_logp_difference/max": 1.524185061454773, "sampling/sampling_logp_difference/mean": 0.009153633378446102, "step": 1973 }, { "clip_ratio/high_max": 0.006263126386329532, "clip_ratio/high_mean": 0.006263126386329532, "clip_ratio/low_mean": 0.007525252411141992, "clip_ratio/low_min": 0.007525252411141992, "clip_ratio/region_mean": 0.013788378797471523, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 100.125, "completions/mean_terminated_length": 100.125, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.10710672754794359, "epoch": 0.07928666104349921, "frac_reward_zero_std": 0.0, "grad_norm": 1.7984775304794312, "learning_rate": 4.0212121212121216e-06, "loss": -0.0005, "num_tokens": 4446120.0, "reward": 0.9985292553901672, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985292553901672, "reward_meter_std": 0.000194300344446674, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00019429647363722324, "reward_total_composite_mean": 0.9985292553901672, "reward_total_composite_std": 0.000194300344446674, "reward_total_mean": 0.9985292553901672, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985292553901672, "rewards/meter/std": 0.000194300344446674, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9985292553901672, "rewards/total_composite/std": 0.000194300344446674, "sampling/importance_sampling_ratio/max": 1.6485373973846436, "sampling/importance_sampling_ratio/mean": 1.0018852949142456, "sampling/importance_sampling_ratio/min": 0.2949223816394806, "sampling/sampling_logp_difference/max": 1.2210431098937988, "sampling/sampling_logp_difference/mean": 0.01695932447910309, "step": 1974 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.011537018930539489, "epoch": 0.07932682652528417, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.018181818181818e-06, "loss": 0.0, "num_tokens": 4447928.0, "reward": 0.9970200061798096, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970200061798096, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9970200061798096, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9970200061798096, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970200061798096, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970200061798096, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.044012427330017, "sampling/importance_sampling_ratio/mean": 1.0012253522872925, "sampling/importance_sampling_ratio/min": 0.9628649353981018, "sampling/sampling_logp_difference/max": 0.04307138919830322, "sampling/sampling_logp_difference/mean": 0.0013655778020620346, "step": 1975 }, { "clip_ratio/high_max": 0.022601958131417632, "clip_ratio/high_mean": 0.022601958131417632, "clip_ratio/low_mean": 0.0077665450517088175, "clip_ratio/low_min": 0.0077665450517088175, "clip_ratio/region_mean": 0.03036850318312645, "completions/clipped_ratio": 0.0, "completions/max_length": 229.0, "completions/max_terminated_length": 229.0, "completions/mean_length": 222.125, "completions/mean_terminated_length": 222.125, "completions/min_length": 216.0, "completions/min_terminated_length": 216.0, "entropy": 0.18782638758420944, "epoch": 0.07936699200706912, "frac_reward_zero_std": 0.0, "grad_norm": 3.668858766555786, "learning_rate": 4.015151515151515e-06, "loss": 0.0074, "num_tokens": 4451433.0, "reward": 0.8963793516159058, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.9412789940834045, "reward_meter_std": 0.15677009522914886, "reward_repeat_penalty_mean": 0.9695513248443604, "reward_repeat_penalty_std": 0.042069755494594574, "reward_std": 0.17684435844421387, "reward_total_composite_mean": 0.8963793516159058, "reward_total_composite_std": 0.17684437334537506, "reward_total_mean": 0.8963793516159058, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.9412789940834045, "rewards/meter/std": 0.15677009522914886, "rewards/repeat_penalty/mean": 0.9695513248443604, "rewards/repeat_penalty/std": 0.042069755494594574, "rewards/total_composite/mean": 0.8963793516159058, "rewards/total_composite/std": 0.17684437334537506, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9989370107650757, "sampling/importance_sampling_ratio/min": 0.0066955070942640305, "sampling/sampling_logp_difference/max": 5.00631856918335, "sampling/sampling_logp_difference/mean": 0.04191061109304428, "step": 1976 }, { "clip_ratio/high_max": 0.02449705172330141, "clip_ratio/high_mean": 0.02449705172330141, "clip_ratio/low_mean": 0.011562932399101555, "clip_ratio/low_min": 0.011562932399101555, "clip_ratio/region_mean": 0.036059984122402966, "completions/clipped_ratio": 0.0, "completions/max_length": 189.0, "completions/max_terminated_length": 189.0, "completions/mean_length": 184.25, "completions/mean_terminated_length": 184.25, "completions/min_length": 179.0, "completions/min_terminated_length": 179.0, "entropy": 0.37936991080641747, "epoch": 0.07940715748885407, "frac_reward_zero_std": 0.0, "grad_norm": 2.525111436843872, "learning_rate": 4.0121212121212125e-06, "loss": 0.0098, "num_tokens": 4454371.0, "reward": 0.9247466325759888, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.993905782699585, "reward_meter_std": 0.00516868568956852, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.08266931027173996, "reward_std": 0.08044657856225967, "reward_total_composite_mean": 0.9247466325759888, "reward_total_composite_std": 0.08044659346342087, "reward_total_mean": 0.9247466325759888, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.993905782699585, "rewards/meter/std": 0.00516868568956852, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.9247466325759888, "rewards/total_composite/std": 0.08044659346342087, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0052932500839233, "sampling/importance_sampling_ratio/min": 0.18207307159900665, "sampling/sampling_logp_difference/max": 1.7033472061157227, "sampling/sampling_logp_difference/mean": 0.049122218042612076, "step": 1977 }, { "clip_ratio/high_max": 0.03106398310046643, "clip_ratio/high_mean": 0.03106398310046643, "clip_ratio/low_mean": 0.009999999776482582, "clip_ratio/low_min": 0.009999999776482582, "clip_ratio/region_mean": 0.04106398287694901, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.25, "completions/mean_terminated_length": 76.25, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.4438166059553623, "epoch": 0.07944732297063903, "frac_reward_zero_std": 0.0, "grad_norm": 5.933767318725586, "learning_rate": 4.009090909090909e-06, "loss": 0.0005, "num_tokens": 4456189.0, "reward": 0.944914698600769, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.944914698600769, "reward_meter_std": 0.14462126791477203, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.14462128281593323, "reward_total_composite_mean": 0.944914698600769, "reward_total_composite_std": 0.14462126791477203, "reward_total_mean": 0.944914698600769, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.944914698600769, "rewards/meter/std": 0.14462126791477203, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.944914698600769, "rewards/total_composite/std": 0.14462126791477203, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005175232887268, "sampling/importance_sampling_ratio/min": 0.2533941864967346, "sampling/sampling_logp_difference/max": 1.3728089332580566, "sampling/sampling_logp_difference/mean": 0.0594903789460659, "step": 1978 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.0056535504991188645, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.03965302975848317, "epoch": 0.07948748845242398, "frac_reward_zero_std": 0.0, "grad_norm": 1.3233561515808105, "learning_rate": 4.006060606060607e-06, "loss": 0.0023, "num_tokens": 4457998.0, "reward": 0.9989874362945557, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989874362945557, "reward_meter_std": 5.0926621042890474e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.0924856623169035e-05, "reward_total_composite_mean": 0.9989874362945557, "reward_total_composite_std": 5.0926621042890474e-05, "reward_total_mean": 0.9989874362945557, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989874362945557, "rewards/meter/std": 5.0926621042890474e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989874362945557, "rewards/total_composite/std": 5.0926621042890474e-05, "sampling/importance_sampling_ratio/max": 1.4895697832107544, "sampling/importance_sampling_ratio/mean": 0.9985904693603516, "sampling/importance_sampling_ratio/min": 0.41957396268844604, "sampling/sampling_logp_difference/max": 0.8685154914855957, "sampling/sampling_logp_difference/mean": 0.010206947103142738, "step": 1979 }, { "clip_ratio/high_max": 0.013719010865315795, "clip_ratio/high_mean": 0.013719010865315795, "clip_ratio/low_mean": 0.010806537233293056, "clip_ratio/low_min": 0.010806537233293056, "clip_ratio/region_mean": 0.02452554809860885, "completions/clipped_ratio": 0.0, "completions/max_length": 222.0, "completions/max_terminated_length": 222.0, "completions/mean_length": 199.75, "completions/mean_terminated_length": 199.75, "completions/min_length": 169.0, "completions/min_terminated_length": 169.0, "entropy": 0.16436910070478916, "epoch": 0.07952765393420894, "frac_reward_zero_std": 0.0, "grad_norm": 2.8739843368530273, "learning_rate": 4.003030303030303e-06, "loss": 0.087, "num_tokens": 4461140.0, "reward": 0.7281157374382019, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_meter_mean": 0.9881318211555481, "reward_meter_std": 0.012296332977712154, "reward_repeat_penalty_mean": 0.8166666626930237, "reward_repeat_penalty_std": 0.05781134217977524, "reward_std": 0.11554723978042603, "reward_total_composite_mean": 0.7281157374382019, "reward_total_composite_std": 0.11554723232984543, "reward_total_mean": 0.7281157374382019, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/meter/mean": 0.9881318211555481, "rewards/meter/std": 0.012296332977712154, "rewards/repeat_penalty/mean": 0.8166666626930237, "rewards/repeat_penalty/std": 0.05781134217977524, "rewards/total_composite/mean": 0.7281157374382019, "rewards/total_composite/std": 0.11554723232984543, "sampling/importance_sampling_ratio/max": 1.922302007675171, "sampling/importance_sampling_ratio/mean": 0.9996204972267151, "sampling/importance_sampling_ratio/min": 0.014542227610945702, "sampling/sampling_logp_difference/max": 4.230698585510254, "sampling/sampling_logp_difference/mean": 0.031819190829992294, "step": 1980 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 90.0, "completions/mean_terminated_length": 90.0, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.011815825360827148, "epoch": 0.07956781941599389, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.000000000000001e-06, "loss": 0.0, "num_tokens": 4463220.0, "reward": 0.9955950975418091, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955950975418091, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9955950975418091, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9955950975418091, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955950975418091, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955950975418091, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.1288472414016724, "sampling/importance_sampling_ratio/mean": 1.0010671615600586, "sampling/importance_sampling_ratio/min": 0.937319278717041, "sampling/sampling_logp_difference/max": 0.12119700014591217, "sampling/sampling_logp_difference/mean": 0.0013981742085888982, "step": 1981 }, { "clip_ratio/high_max": 0.01946915965527296, "clip_ratio/high_mean": 0.01946915965527296, "clip_ratio/low_mean": 0.008034686907194555, "clip_ratio/low_min": 0.008034686907194555, "clip_ratio/region_mean": 0.027503846562467515, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.875, "completions/mean_terminated_length": 76.875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.357822060585022, "epoch": 0.07960798489777884, "frac_reward_zero_std": 0.0, "grad_norm": 6.157790660858154, "learning_rate": 3.996969696969698e-06, "loss": 0.0058, "num_tokens": 4465131.0, "reward": 0.9915469884872437, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9915469884872437, "reward_meter_std": 0.0046110316179692745, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004611044656485319, "reward_total_composite_mean": 0.9915469884872437, "reward_total_composite_std": 0.0046110316179692745, "reward_total_mean": 0.9915469884872437, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9915469884872437, "rewards/meter/std": 0.0046110316179692745, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9915469884872437, "rewards/total_composite/std": 0.0046110316179692745, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0080691576004028, "sampling/importance_sampling_ratio/min": 0.3472263514995575, "sampling/sampling_logp_difference/max": 1.0577783584594727, "sampling/sampling_logp_difference/mean": 0.03136689215898514, "step": 1982 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.00390625, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.875, "completions/mean_terminated_length": 32.875, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.06188688660040498, "epoch": 0.0796481503795638, "frac_reward_zero_std": 0.0, "grad_norm": 4.942656993865967, "learning_rate": 3.993939393939394e-06, "loss": -0.0135, "num_tokens": 4466706.0, "reward": 0.9034555554389954, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9034555554389954, "reward_meter_std": 0.1775951385498047, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1775951385498047, "reward_total_composite_mean": 0.9034555554389954, "reward_total_composite_std": 0.1775951385498047, "reward_total_mean": 0.9034555554389954, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9034555554389954, "rewards/meter/std": 0.1775951385498047, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9034555554389954, "rewards/total_composite/std": 0.1775951385498047, "sampling/importance_sampling_ratio/max": 1.1638069152832031, "sampling/importance_sampling_ratio/mean": 1.0033267736434937, "sampling/importance_sampling_ratio/min": 0.6683841943740845, "sampling/sampling_logp_difference/max": 0.4028921127319336, "sampling/sampling_logp_difference/mean": 0.0072293514385819435, "step": 1983 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.003703906899318099, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.03432013071142137, "epoch": 0.07968831586134875, "frac_reward_zero_std": 0.0, "grad_norm": 0.42240381240844727, "learning_rate": 3.990909090909092e-06, "loss": -0.0081, "num_tokens": 4468579.0, "reward": 0.9162052869796753, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9162052869796753, "reward_meter_std": 0.0074835000559687614, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007483497262001038, "reward_total_composite_mean": 0.9162052869796753, "reward_total_composite_std": 0.0074835000559687614, "reward_total_mean": 0.9162052869796753, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9162052869796753, "rewards/meter/std": 0.0074835000559687614, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9162052869796753, "rewards/total_composite/std": 0.0074835000559687614, "sampling/importance_sampling_ratio/max": 1.473544955253601, "sampling/importance_sampling_ratio/mean": 1.0025385618209839, "sampling/importance_sampling_ratio/min": 0.7210829854011536, "sampling/sampling_logp_difference/max": 0.38767099380493164, "sampling/sampling_logp_difference/mean": 0.004512776155024767, "step": 1984 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.007963738986290991, "epoch": 0.0797284813431337, "frac_reward_zero_std": 0.0, "grad_norm": 1.1522923707962036, "learning_rate": 3.987878787878788e-06, "loss": 0.0002, "num_tokens": 4470379.0, "reward": 0.994735836982727, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994735836982727, "reward_meter_std": 6.81514575262554e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.814543303335086e-05, "reward_total_composite_mean": 0.994735836982727, "reward_total_composite_std": 6.81514575262554e-05, "reward_total_mean": 0.994735836982727, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994735836982727, "rewards/meter/std": 6.81514575262554e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994735836982727, "rewards/total_composite/std": 6.81514575262554e-05, "sampling/importance_sampling_ratio/max": 1.0196856260299683, "sampling/importance_sampling_ratio/mean": 1.0006917715072632, "sampling/importance_sampling_ratio/min": 0.9891203045845032, "sampling/sampling_logp_difference/max": 0.019494354724884033, "sampling/sampling_logp_difference/mean": 0.0007778764702379704, "step": 1985 }, { "clip_ratio/high_max": 0.03596452111378312, "clip_ratio/high_mean": 0.03596452111378312, "clip_ratio/low_mean": 0.05110658518970013, "clip_ratio/low_min": 0.05110658518970013, "clip_ratio/region_mean": 0.08707110630348325, "completions/clipped_ratio": 0.0, "completions/max_length": 43.0, "completions/max_terminated_length": 43.0, "completions/mean_length": 40.625, "completions/mean_terminated_length": 40.625, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.6491754315793514, "epoch": 0.07976864682491866, "frac_reward_zero_std": 0.0, "grad_norm": 13.343822479248047, "learning_rate": 3.984848484848485e-06, "loss": 0.0087, "num_tokens": 4471880.0, "reward": 0.47001466155052185, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5094391703605652, "reward_meter_std": 0.4160431921482086, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.39801424741744995, "reward_total_composite_mean": 0.47001466155052185, "reward_total_composite_std": 0.39801424741744995, "reward_total_mean": 0.47001466155052185, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5094391703605652, "rewards/meter/std": 0.4160431921482086, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.47001466155052185, "rewards/total_composite/std": 0.39801424741744995, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0084450244903564, "sampling/importance_sampling_ratio/min": 0.22063790261745453, "sampling/sampling_logp_difference/max": 1.5112323760986328, "sampling/sampling_logp_difference/mean": 0.10109325498342514, "step": 1986 }, { "clip_ratio/high_max": 0.02709868340753019, "clip_ratio/high_mean": 0.02709868340753019, "clip_ratio/low_mean": 0.010287561686709523, "clip_ratio/low_min": 0.010287561686709523, "clip_ratio/region_mean": 0.03738624509423971, "completions/clipped_ratio": 0.0, "completions/max_length": 156.0, "completions/max_terminated_length": 156.0, "completions/mean_length": 149.375, "completions/mean_terminated_length": 149.375, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.4626317396759987, "epoch": 0.07980881230670361, "frac_reward_zero_std": 0.0, "grad_norm": 4.379970073699951, "learning_rate": 3.9818181818181825e-06, "loss": -0.0103, "num_tokens": 4474379.0, "reward": 0.9405426383018494, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0883883461356163, "reward_meter_mean": 0.9875733852386475, "reward_meter_std": 0.02084759995341301, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.10836568474769592, "reward_total_composite_mean": 0.9405426383018494, "reward_total_composite_std": 0.10836569964885712, "reward_total_mean": 0.9405426383018494, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0883883461356163, "rewards/meter/mean": 0.9875733852386475, "rewards/meter/std": 0.02084759995341301, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9405426383018494, "rewards/total_composite/std": 0.10836569964885712, "sampling/importance_sampling_ratio/max": 1.912219524383545, "sampling/importance_sampling_ratio/mean": 1.0045157670974731, "sampling/importance_sampling_ratio/min": 0.18473687767982483, "sampling/sampling_logp_difference/max": 1.6888227462768555, "sampling/sampling_logp_difference/mean": 0.04869993403553963, "step": 1987 }, { "clip_ratio/high_max": 0.022676826687529683, "clip_ratio/high_mean": 0.022676826687529683, "clip_ratio/low_mean": 0.008121267077513039, "clip_ratio/low_min": 0.008121267077513039, "clip_ratio/region_mean": 0.030798093765042722, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.875, "completions/mean_terminated_length": 76.875, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.4746681675314903, "epoch": 0.07984897778848857, "frac_reward_zero_std": 0.0, "grad_norm": 6.39749002456665, "learning_rate": 3.978787878787879e-06, "loss": 0.0169, "num_tokens": 4476330.0, "reward": 0.988801121711731, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.988801121711731, "reward_meter_std": 0.014979206025600433, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014979207888245583, "reward_total_composite_mean": 0.988801121711731, "reward_total_composite_std": 0.014979206025600433, "reward_total_mean": 0.988801121711731, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.988801121711731, "rewards/meter/std": 0.014979206025600433, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.988801121711731, "rewards/total_composite/std": 0.014979206025600433, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0134505033493042, "sampling/importance_sampling_ratio/min": 0.21401913464069366, "sampling/sampling_logp_difference/max": 1.5416898727416992, "sampling/sampling_logp_difference/mean": 0.05862836539745331, "step": 1988 }, { "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/low_mean": 0.005597014795057476, "clip_ratio/low_min": 0.005597014795057476, "clip_ratio/region_mean": 0.00927348539698869, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.5, "completions/mean_terminated_length": 67.5, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.08089068345725536, "epoch": 0.07988914327027352, "frac_reward_zero_std": 0.0, "grad_norm": 5.730335712432861, "learning_rate": 3.975757575757576e-06, "loss": -0.0063, "num_tokens": 4478142.0, "reward": 0.9175406694412231, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9175406694412231, "reward_meter_std": 0.021055592224001884, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.021055590361356735, "reward_total_composite_mean": 0.9175406694412231, "reward_total_composite_std": 0.021055592224001884, "reward_total_mean": 0.9175406694412231, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9175406694412231, "rewards/meter/std": 0.021055592224001884, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9175406694412231, "rewards/total_composite/std": 0.021055592224001884, "sampling/importance_sampling_ratio/max": 1.4384418725967407, "sampling/importance_sampling_ratio/mean": 1.0017634630203247, "sampling/importance_sampling_ratio/min": 0.38260820508003235, "sampling/sampling_logp_difference/max": 0.96074378490448, "sampling/sampling_logp_difference/mean": 0.014552988111972809, "step": 1989 }, { "clip_ratio/high_max": 0.016820423072203994, "clip_ratio/high_mean": 0.016820423072203994, "clip_ratio/low_mean": 0.005168477655388415, "clip_ratio/low_min": 0.005168477655388415, "clip_ratio/region_mean": 0.02198890072759241, "completions/clipped_ratio": 0.0, "completions/max_length": 222.0, "completions/max_terminated_length": 222.0, "completions/mean_length": 216.125, "completions/mean_terminated_length": 216.125, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.09882919304072857, "epoch": 0.07992930875205848, "frac_reward_zero_std": 0.0, "grad_norm": 2.3331921100616455, "learning_rate": 3.972727272727273e-06, "loss": -0.0004, "num_tokens": 4481575.0, "reward": 0.9027365446090698, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998195230960846, "reward_meter_std": 0.0007794296252541244, "reward_repeat_penalty_mean": 0.9043561220169067, "reward_repeat_penalty_std": 0.053097911179065704, "reward_std": 0.05324570834636688, "reward_total_composite_mean": 0.9027365446090698, "reward_total_composite_std": 0.053245700895786285, "reward_total_mean": 0.9027365446090698, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998195230960846, "rewards/meter/std": 0.0007794296252541244, "rewards/repeat_penalty/mean": 0.9043561220169067, "rewards/repeat_penalty/std": 0.053097911179065704, "rewards/total_composite/mean": 0.9027365446090698, "rewards/total_composite/std": 0.053245700895786285, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023928880691528, "sampling/importance_sampling_ratio/min": 0.21839307248592377, "sampling/sampling_logp_difference/max": 1.5214588642120361, "sampling/sampling_logp_difference/mean": 0.023263096809387207, "step": 1990 }, { "clip_ratio/high_max": 0.007373977248789743, "clip_ratio/high_mean": 0.007373977248789743, "clip_ratio/low_mean": 0.001727131224470213, "clip_ratio/low_min": 0.001727131224470213, "clip_ratio/region_mean": 0.009101108473259956, "completions/clipped_ratio": 0.0, "completions/max_length": 290.0, "completions/max_terminated_length": 290.0, "completions/mean_length": 288.375, "completions/mean_terminated_length": 288.375, "completions/min_length": 287.0, "completions/min_terminated_length": 287.0, "entropy": 0.059712667018175125, "epoch": 0.07996947423384343, "frac_reward_zero_std": 0.0, "grad_norm": 1.446068286895752, "learning_rate": 3.96969696969697e-06, "loss": 0.0014, "num_tokens": 4485554.0, "reward": 0.411979079246521, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9093496203422546, "reward_meter_std": 0.01770671270787716, "reward_repeat_penalty_mean": 0.6796875, "reward_repeat_penalty_std": 0.022097086533904076, "reward_std": 0.013528193347156048, "reward_total_composite_mean": 0.411979079246521, "reward_total_composite_std": 0.013528194278478622, "reward_total_mean": 0.411979079246521, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9093496203422546, "rewards/meter/std": 0.01770671270787716, "rewards/repeat_penalty/mean": 0.6796875, "rewards/repeat_penalty/std": 0.022097086533904076, "rewards/total_composite/mean": 0.411979079246521, "rewards/total_composite/std": 0.013528194278478622, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9994128942489624, "sampling/importance_sampling_ratio/min": 9.368746418658702e-07, "sampling/sampling_logp_difference/max": 13.880716323852539, "sampling/sampling_logp_difference/mean": 0.02342934161424637, "step": 1991 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 159.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 157.875, "completions/mean_terminated_length": 157.875, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "entropy": 0.011172075115609914, "epoch": 0.08000963971562838, "frac_reward_zero_std": 0.0, "grad_norm": 1.818966031074524, "learning_rate": 3.966666666666667e-06, "loss": -0.0152, "num_tokens": 4488257.0, "reward": 0.7754957675933838, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970660209655762, "reward_meter_std": 0.00133254355750978, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001036411034874618, "reward_total_composite_mean": 0.7754957675933838, "reward_total_composite_std": 0.0010364169720560312, "reward_total_mean": 0.7754957675933838, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970660209655762, "rewards/meter/std": 0.00133254355750978, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7754957675933838, "rewards/total_composite/std": 0.0010364169720560312, "sampling/importance_sampling_ratio/max": 1.1387183666229248, "sampling/importance_sampling_ratio/mean": 1.0000977516174316, "sampling/importance_sampling_ratio/min": 0.44438278675079346, "sampling/sampling_logp_difference/max": 0.8110690116882324, "sampling/sampling_logp_difference/mean": 0.002477371832355857, "step": 1992 }, { "clip_ratio/high_max": 0.011917525203898549, "clip_ratio/high_mean": 0.011917525203898549, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/region_mean": 0.015389747451990843, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 73.0, "completions/mean_terminated_length": 73.0, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.21745091304183006, "epoch": 0.08004980519741334, "frac_reward_zero_std": 0.0, "grad_norm": 3.636791467666626, "learning_rate": 3.963636363636364e-06, "loss": -0.0053, "num_tokens": 4490233.0, "reward": 0.9986278414726257, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986278414726257, "reward_meter_std": 0.0007868955726735294, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007868955144658685, "reward_total_composite_mean": 0.9986278414726257, "reward_total_composite_std": 0.0007868955726735294, "reward_total_mean": 0.9986278414726257, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986278414726257, "rewards/meter/std": 0.0007868955726735294, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986278414726257, "rewards/total_composite/std": 0.0007868955726735294, "sampling/importance_sampling_ratio/max": 1.5565049648284912, "sampling/importance_sampling_ratio/mean": 1.0071430206298828, "sampling/importance_sampling_ratio/min": 0.3592449724674225, "sampling/sampling_logp_difference/max": 1.0237507820129395, "sampling/sampling_logp_difference/mean": 0.03113717958331108, "step": 1993 }, { "clip_ratio/high_max": 0.03634089510887861, "clip_ratio/high_mean": 0.03634089510887861, "clip_ratio/low_mean": 0.006076388992369175, "clip_ratio/low_min": 0.006076388992369175, "clip_ratio/region_mean": 0.04241728410124779, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 147.375, "completions/mean_terminated_length": 147.375, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.4670833945274353, "epoch": 0.08008997067919829, "frac_reward_zero_std": 0.0, "grad_norm": 3.498656749725342, "learning_rate": 3.960606060606061e-06, "loss": -0.0042, "num_tokens": 4492892.0, "reward": 0.9948342442512512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948342442512512, "reward_meter_std": 0.006030126474797726, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006030108779668808, "reward_total_composite_mean": 0.9948342442512512, "reward_total_composite_std": 0.006030126474797726, "reward_total_mean": 0.9948342442512512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948342442512512, "rewards/meter/std": 0.006030126474797726, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948342442512512, "rewards/total_composite/std": 0.006030126474797726, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0064527988433838, "sampling/importance_sampling_ratio/min": 0.16218742728233337, "sampling/sampling_logp_difference/max": 1.819002628326416, "sampling/sampling_logp_difference/mean": 0.06232795864343643, "step": 1994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.012096773833036423, "clip_ratio/low_min": 0.012096773833036423, "clip_ratio/region_mean": 0.012096773833036423, "completions/clipped_ratio": 0.0, "completions/max_length": 31.0, "completions/max_terminated_length": 31.0, "completions/mean_length": 31.0, "completions/mean_terminated_length": 31.0, "completions/min_length": 31.0, "completions/min_terminated_length": 31.0, "entropy": 0.03334390290547162, "epoch": 0.08013013616098325, "frac_reward_zero_std": 0.0, "grad_norm": 0.7663399577140808, "learning_rate": 3.957575757575758e-06, "loss": 0.0012, "num_tokens": 4494276.0, "reward": 0.9957044124603271, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957044124603271, "reward_meter_std": 3.0345732739078812e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.034573092008941e-05, "reward_total_composite_mean": 0.9957044124603271, "reward_total_composite_std": 3.0345732739078812e-05, "reward_total_mean": 0.9957044124603271, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957044124603271, "rewards/meter/std": 3.0345732739078812e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957044124603271, "rewards/total_composite/std": 3.0345732739078812e-05, "sampling/importance_sampling_ratio/max": 1.0392725467681885, "sampling/importance_sampling_ratio/mean": 0.9947730898857117, "sampling/importance_sampling_ratio/min": 0.01784687116742134, "sampling/sampling_logp_difference/max": 4.0259270668029785, "sampling/sampling_logp_difference/mean": 0.02179253287613392, "step": 1995 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.011029411805793643, "clip_ratio/low_min": 0.011029411805793643, "clip_ratio/region_mean": 0.011029411805793643, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 34.25, "completions/mean_terminated_length": 34.25, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.046473859809339046, "epoch": 0.0801703016427682, "frac_reward_zero_std": 0.0, "grad_norm": 11.98099422454834, "learning_rate": 3.954545454545454e-06, "loss": -0.0084, "num_tokens": 4495750.0, "reward": 0.9950810074806213, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9950810074806213, "reward_meter_std": 0.0017581552965566516, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0017581689171493053, "reward_total_composite_mean": 0.9950810074806213, "reward_total_composite_std": 0.0017581552965566516, "reward_total_mean": 0.9950810074806213, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9950810074806213, "rewards/meter/std": 0.0017581552965566516, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950810074806213, "rewards/total_composite/std": 0.0017581552965566516, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008811354637146, "sampling/importance_sampling_ratio/min": 0.549746036529541, "sampling/sampling_logp_difference/max": 1.0786250829696655, "sampling/sampling_logp_difference/mean": 0.015304782427847385, "step": 1996 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.048666974529623985, "epoch": 0.08021046712455315, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.951515151515152e-06, "loss": 0.0, "num_tokens": 4497582.0, "reward": 0.9355690479278564, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9355690479278564, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9355690479278564, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9355690479278564, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9355690479278564, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9355690479278564, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.093409538269043, "sampling/importance_sampling_ratio/mean": 1.004404067993164, "sampling/importance_sampling_ratio/min": 0.9659483432769775, "sampling/sampling_logp_difference/max": 0.08930085599422455, "sampling/sampling_logp_difference/mean": 0.0048114219680428505, "step": 1997 }, { "clip_ratio/high_max": 0.004166666883975267, "clip_ratio/high_mean": 0.004166666883975267, "clip_ratio/low_mean": 0.01666666753590107, "clip_ratio/low_min": 0.01666666753590107, "clip_ratio/region_mean": 0.020833334419876337, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 30.0, "completions/mean_terminated_length": 30.0, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.05545296333730221, "epoch": 0.08025063260633811, "frac_reward_zero_std": 0.0, "grad_norm": 1.003632664680481, "learning_rate": 3.948484848484849e-06, "loss": 0.0006, "num_tokens": 4499102.0, "reward": 0.9927566647529602, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9927566647529602, "reward_meter_std": 3.355137232574634e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.355137232574634e-05, "reward_total_composite_mean": 0.9927566647529602, "reward_total_composite_std": 3.355137232574634e-05, "reward_total_mean": 0.9927566647529602, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9927566647529602, "rewards/meter/std": 3.355137232574634e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9927566647529602, "rewards/total_composite/std": 3.355137232574634e-05, "sampling/importance_sampling_ratio/max": 1.4227073192596436, "sampling/importance_sampling_ratio/mean": 1.0059435367584229, "sampling/importance_sampling_ratio/min": 0.6398874521255493, "sampling/sampling_logp_difference/max": 0.4464629590511322, "sampling/sampling_logp_difference/mean": 0.010052908211946487, "step": 1998 }, { "clip_ratio/high_max": 0.02339910331647843, "clip_ratio/high_mean": 0.02339910331647843, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/region_mean": 0.03196074708830565, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 73.875, "completions/mean_terminated_length": 73.875, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.40391664765775204, "epoch": 0.08029079808812306, "frac_reward_zero_std": 0.0, "grad_norm": 2.3003454208374023, "learning_rate": 3.945454545454545e-06, "loss": -0.0049, "num_tokens": 4501021.0, "reward": 0.9951367378234863, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9951367378234863, "reward_meter_std": 0.003929613158106804, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003929623868316412, "reward_total_composite_mean": 0.9951367378234863, "reward_total_composite_std": 0.003929613158106804, "reward_total_mean": 0.9951367378234863, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9951367378234863, "rewards/meter/std": 0.003929613158106804, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9951367378234863, "rewards/total_composite/std": 0.003929613158106804, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0134656429290771, "sampling/importance_sampling_ratio/min": 0.22839628159999847, "sampling/sampling_logp_difference/max": 1.4766731262207031, "sampling/sampling_logp_difference/mean": 0.04188167303800583, "step": 1999 }, { "clip_ratio/high_max": 0.026401946786791086, "clip_ratio/high_mean": 0.026401946786791086, "clip_ratio/low_mean": 0.005059222981799394, "clip_ratio/low_min": 0.005059222981799394, "clip_ratio/region_mean": 0.03146116976859048, "completions/clipped_ratio": 0.0, "completions/max_length": 153.0, "completions/max_terminated_length": 153.0, "completions/mean_length": 147.125, "completions/mean_terminated_length": 147.125, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 0.3696784619241953, "epoch": 0.08033096356990801, "frac_reward_zero_std": 0.0, "grad_norm": 3.8840928077697754, "learning_rate": 3.942424242424243e-06, "loss": 0.0069, "num_tokens": 4503974.0, "reward": 0.9359963536262512, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9887276887893677, "reward_meter_std": 0.01330604124814272, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07730759680271149, "reward_total_composite_mean": 0.9359963536262512, "reward_total_composite_std": 0.07730759680271149, "reward_total_mean": 0.9359963536262512, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9887276887893677, "rewards/meter/std": 0.01330604124814272, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9359963536262512, "rewards/total_composite/std": 0.07730759680271149, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0066624879837036, "sampling/importance_sampling_ratio/min": 0.06833374500274658, "sampling/sampling_logp_difference/max": 2.683351516723633, "sampling/sampling_logp_difference/mean": 0.05714353546500206, "step": 2000 }, { "epoch": 0.08033096356990801, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 358.15384615384613, "eval_completions/max_terminated_length": 358.15384615384613, "eval_completions/mean_length": 205.2596153846154, "eval_completions/mean_terminated_length": 205.2596153846154, "eval_completions/min_length": 62.92307692307692, "eval_completions/min_terminated_length": 62.92307692307692, "eval_entropy": 0.16280974046542093, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 4503974.0, "eval_reward": 0.568704937513058, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9229050003565274, "eval_reward_count_adherence_std": 0.11920174153951499, "eval_reward_meter_mean": 0.7367691305967478, "eval_reward_meter_std": 0.3722034446322001, "eval_reward_repeat_penalty_mean": 0.8324548143606919, "eval_reward_repeat_penalty_std": 0.1328287388269718, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.568704937513058, "eval_reward_total_composite_std": 0.34218758459274584, "eval_reward_total_mean": 0.568704937513058, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9229050003565274, "eval_rewards/count_adherence/std": 0.11920174153951499, "eval_rewards/meter/mean": 0.7367691305967478, "eval_rewards/meter/std": 0.3722034446322001, "eval_rewards/repeat_penalty/mean": 0.8324548143606919, "eval_rewards/repeat_penalty/std": 0.1328287388269718, "eval_rewards/total_composite/mean": 0.568704937513058, "eval_rewards/total_composite/std": 0.34218758459274584, "eval_runtime": 67.8915, "eval_samples_per_second": 1.532, "eval_sampling/importance_sampling_ratio/max": 1.4441173993624175, "eval_sampling/importance_sampling_ratio/mean": 1.0038928618797889, "eval_sampling/importance_sampling_ratio/min": 0.33040788540473354, "eval_sampling/sampling_logp_difference/max": 1.1488754015702467, "eval_sampling/sampling_logp_difference/mean": 0.017686504655732557, "eval_steps_per_second": 0.191, "step": 2000 }, { "clip_ratio/high_max": 0.0010245901066809893, "clip_ratio/high_mean": 0.0010245901066809893, "clip_ratio/low_mean": 0.005122950533404946, "clip_ratio/low_min": 0.005122950533404946, "clip_ratio/region_mean": 0.006147540640085936, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 122.0, "completions/mean_terminated_length": 122.0, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.043996233493089676, "epoch": 0.08037112905169297, "frac_reward_zero_std": 0.0, "grad_norm": 1.4718478918075562, "learning_rate": 3.93939393939394e-06, "loss": 0.0013, "num_tokens": 4506302.0, "reward": 0.8717328310012817, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9962645173072815, "reward_meter_std": 4.681191057898104e-05, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05034938082098961, "reward_total_composite_mean": 0.8717328310012817, "reward_total_composite_std": 0.050349388271570206, "reward_total_mean": 0.8717328310012817, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9962645173072815, "rewards/meter/std": 4.681191057898104e-05, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8717328310012817, "rewards/total_composite/std": 0.050349388271570206, "sampling/importance_sampling_ratio/max": 1.2980602979660034, "sampling/importance_sampling_ratio/mean": 1.0027551651000977, "sampling/importance_sampling_ratio/min": 0.30166009068489075, "sampling/sampling_logp_difference/max": 1.1984543800354004, "sampling/sampling_logp_difference/mean": 0.007403408642858267, "step": 2001 }, { "clip_ratio/high_max": 0.0032051282469183207, "clip_ratio/high_mean": 0.0032051282469183207, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0032051282469183207, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 37.375, "completions/mean_terminated_length": 37.375, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.1442318456247449, "epoch": 0.08041129453347792, "frac_reward_zero_std": 0.0, "grad_norm": 5.45490837097168, "learning_rate": 3.936363636363636e-06, "loss": -0.0194, "num_tokens": 4507873.0, "reward": 0.9985835552215576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985835552215576, "reward_meter_std": 0.00047840786282904446, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004784138291142881, "reward_total_composite_mean": 0.9985835552215576, "reward_total_composite_std": 0.00047840786282904446, "reward_total_mean": 0.9985835552215576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985835552215576, "rewards/meter/std": 0.00047840786282904446, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9985835552215576, "rewards/total_composite/std": 0.00047840786282904446, "sampling/importance_sampling_ratio/max": 1.2333381175994873, "sampling/importance_sampling_ratio/mean": 1.0080413818359375, "sampling/importance_sampling_ratio/min": 0.49037110805511475, "sampling/sampling_logp_difference/max": 0.7125928401947021, "sampling/sampling_logp_difference/mean": 0.017835740000009537, "step": 2002 }, { "clip_ratio/high_max": 0.0009842519648373127, "clip_ratio/high_mean": 0.0009842519648373127, "clip_ratio/low_mean": 0.0009842519648373127, "clip_ratio/low_min": 0.0009842519648373127, "clip_ratio/region_mean": 0.0019685039296746254, "completions/clipped_ratio": 0.0, "completions/max_length": 127.0, "completions/max_terminated_length": 127.0, "completions/mean_length": 127.0, "completions/mean_terminated_length": 127.0, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.007686095021199435, "epoch": 0.08045146001526288, "frac_reward_zero_std": 0.0, "grad_norm": 0.0041707647033035755, "learning_rate": 3.9333333333333335e-06, "loss": -0.0002, "num_tokens": 4510281.0, "reward": 0.854958176612854, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974512457847595, "reward_meter_std": 1.9482802599668503e-05, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 1.669656739977654e-05, "reward_total_composite_mean": 0.854958176612854, "reward_total_composite_std": 1.66965983225964e-05, "reward_total_mean": 0.854958176612854, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974512457847595, "rewards/meter/std": 1.9482802599668503e-05, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.854958176612854, "rewards/total_composite/std": 1.66965983225964e-05, "sampling/importance_sampling_ratio/max": 1.0359206199645996, "sampling/importance_sampling_ratio/mean": 1.000118374824524, "sampling/importance_sampling_ratio/min": 0.2997877895832062, "sampling/sampling_logp_difference/max": 1.2046804428100586, "sampling/sampling_logp_difference/mean": 0.0020607816986739635, "step": 2003 }, { "clip_ratio/high_max": 0.03295032726600766, "clip_ratio/high_mean": 0.03295032726600766, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.03295032726600766, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 38.125, "completions/mean_terminated_length": 38.125, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.2946350034326315, "epoch": 0.08049162549704783, "frac_reward_zero_std": 0.0, "grad_norm": 4.7584662437438965, "learning_rate": 3.930303030303031e-06, "loss": 0.0022, "num_tokens": 4511890.0, "reward": 0.8744252920150757, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999361515045166, "reward_meter_std": 0.00022483796055894345, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35332125425338745, "reward_total_composite_mean": 0.8744252920150757, "reward_total_composite_std": 0.35332125425338745, "reward_total_mean": 0.8744252920150757, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999361515045166, "rewards/meter/std": 0.00022483796055894345, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8744252920150757, "rewards/total_composite/std": 0.35332125425338745, "sampling/importance_sampling_ratio/max": 1.5466128587722778, "sampling/importance_sampling_ratio/mean": 1.0098013877868652, "sampling/importance_sampling_ratio/min": 0.273834764957428, "sampling/sampling_logp_difference/max": 1.2952303886413574, "sampling/sampling_logp_difference/mean": 0.044290993362665176, "step": 2004 }, { "clip_ratio/high_max": 0.0072237912099808455, "clip_ratio/high_mean": 0.0072237912099808455, "clip_ratio/low_mean": 0.00512295076623559, "clip_ratio/low_min": 0.00512295076623559, "clip_ratio/region_mean": 0.012346741976216435, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 121.625, "completions/mean_terminated_length": 121.625, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.07582355523481965, "epoch": 0.08053179097883278, "frac_reward_zero_std": 0.0, "grad_norm": 2.326425552368164, "learning_rate": 3.927272727272727e-06, "loss": 0.0085, "num_tokens": 4514367.0, "reward": 0.8892085552215576, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959496259689331, "reward_meter_std": 0.0007494749734178185, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06532733887434006, "reward_total_composite_mean": 0.8892085552215576, "reward_total_composite_std": 0.06532733142375946, "reward_total_mean": 0.8892085552215576, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959496259689331, "rewards/meter/std": 0.0007494749734178185, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8892085552215576, "rewards/total_composite/std": 0.06532733142375946, "sampling/importance_sampling_ratio/max": 1.928596019744873, "sampling/importance_sampling_ratio/mean": 1.0025670528411865, "sampling/importance_sampling_ratio/min": 0.27907872200012207, "sampling/sampling_logp_difference/max": 1.276261329650879, "sampling/sampling_logp_difference/mean": 0.011545258574187756, "step": 2005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.04299367079511285, "epoch": 0.08057195646061774, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.9242424242424244e-06, "loss": 0.0, "num_tokens": 4515983.0, "reward": 0.9979252815246582, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979252815246582, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9979252815246582, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9979252815246582, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979252815246582, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979252815246582, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.1792465448379517, "sampling/importance_sampling_ratio/mean": 1.0037531852722168, "sampling/importance_sampling_ratio/min": 0.9554311633110046, "sampling/sampling_logp_difference/max": 0.16487574577331543, "sampling/sampling_logp_difference/mean": 0.004444928839802742, "step": 2006 }, { "clip_ratio/high_max": 0.026706379372626543, "clip_ratio/high_mean": 0.026706379372626543, "clip_ratio/low_mean": 0.00826923071872443, "clip_ratio/low_min": 0.00826923071872443, "clip_ratio/region_mean": 0.03497561009135097, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 74.75, "completions/mean_terminated_length": 74.75, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.378201762214303, "epoch": 0.08061212194240269, "frac_reward_zero_std": 0.0, "grad_norm": 3.622284412384033, "learning_rate": 3.921212121212122e-06, "loss": 0.0055, "num_tokens": 4518013.0, "reward": 0.9929636716842651, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9929636716842651, "reward_meter_std": 0.008374504745006561, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008374514989554882, "reward_total_composite_mean": 0.9929636716842651, "reward_total_composite_std": 0.008374504745006561, "reward_total_mean": 0.9929636716842651, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9929636716842651, "rewards/meter/std": 0.008374504745006561, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929636716842651, "rewards/total_composite/std": 0.008374504745006561, "sampling/importance_sampling_ratio/max": 1.6419928073883057, "sampling/importance_sampling_ratio/mean": 1.006500244140625, "sampling/importance_sampling_ratio/min": 0.30303072929382324, "sampling/sampling_logp_difference/max": 1.1939210891723633, "sampling/sampling_logp_difference/mean": 0.04201868921518326, "step": 2007 }, { "clip_ratio/high_max": 0.007035914051812142, "clip_ratio/high_mean": 0.007035914051812142, "clip_ratio/low_mean": 0.009641719050705433, "clip_ratio/low_min": 0.009641719050705433, "clip_ratio/region_mean": 0.016677633102517575, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 324.75, "completions/mean_terminated_length": 324.75, "completions/min_length": 309.0, "completions/min_terminated_length": 309.0, "entropy": 0.1241519721224904, "epoch": 0.08065228742418765, "frac_reward_zero_std": 0.0, "grad_norm": 1.4125186204910278, "learning_rate": 3.918181818181819e-06, "loss": 0.0072, "num_tokens": 4522347.0, "reward": 0.38363176584243774, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7302596569061279, "reward_meter_std": 0.25050464272499084, "reward_repeat_penalty_mean": 0.7244791984558105, "reward_repeat_penalty_std": 0.03113570623099804, "reward_std": 0.13244113326072693, "reward_total_composite_mean": 0.38363176584243774, "reward_total_composite_std": 0.13244113326072693, "reward_total_mean": 0.38363176584243774, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7302596569061279, "rewards/meter/std": 0.25050464272499084, "rewards/repeat_penalty/mean": 0.7244791984558105, "rewards/repeat_penalty/std": 0.03113570623099804, "rewards/total_composite/mean": 0.38363176584243774, "rewards/total_composite/std": 0.13244113326072693, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0025417804718018, "sampling/importance_sampling_ratio/min": 1.5065788829815574e-06, "sampling/sampling_logp_difference/max": 13.405669212341309, "sampling/sampling_logp_difference/mean": 0.03116171434521675, "step": 2008 }, { "clip_ratio/high_max": 0.006465517217293382, "clip_ratio/high_mean": 0.006465517217293382, "clip_ratio/low_mean": 0.0022321429569274187, "clip_ratio/low_min": 0.0022321429569274187, "clip_ratio/region_mean": 0.0086976601742208, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 57.625, "completions/mean_terminated_length": 57.625, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.08374510239809752, "epoch": 0.0806924529059726, "frac_reward_zero_std": 0.0, "grad_norm": 2.9392364025115967, "learning_rate": 3.915151515151515e-06, "loss": -0.0129, "num_tokens": 4524064.0, "reward": 0.9944944977760315, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9944944977760315, "reward_meter_std": 0.0007702266448177397, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007702277507632971, "reward_total_composite_mean": 0.9944944977760315, "reward_total_composite_std": 0.0007702266448177397, "reward_total_mean": 0.9944944977760315, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9944944977760315, "rewards/meter/std": 0.0007702266448177397, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944944977760315, "rewards/total_composite/std": 0.0007702266448177397, "sampling/importance_sampling_ratio/max": 1.263137936592102, "sampling/importance_sampling_ratio/mean": 1.0033068656921387, "sampling/importance_sampling_ratio/min": 0.5953003764152527, "sampling/sampling_logp_difference/max": 0.5186891555786133, "sampling/sampling_logp_difference/mean": 0.011414974927902222, "step": 2009 }, { "clip_ratio/high_max": 0.00849514571018517, "clip_ratio/high_mean": 0.00849514571018517, "clip_ratio/low_mean": 0.005918904324062169, "clip_ratio/low_min": 0.005918904324062169, "clip_ratio/region_mean": 0.014414050034247339, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 103.5, "completions/mean_terminated_length": 103.5, "completions/min_length": 103.0, "completions/min_terminated_length": 103.0, "entropy": 0.08177287224680185, "epoch": 0.08073261838775755, "frac_reward_zero_std": 0.0, "grad_norm": 2.544851303100586, "learning_rate": 3.912121212121213e-06, "loss": 0.008, "num_tokens": 4526204.0, "reward": 0.8996114730834961, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8996114730834961, "reward_meter_std": 0.042233821004629135, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04223383218050003, "reward_total_composite_mean": 0.8996114730834961, "reward_total_composite_std": 0.042233821004629135, "reward_total_mean": 0.8996114730834961, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8996114730834961, "rewards/meter/std": 0.042233821004629135, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8996114730834961, "rewards/total_composite/std": 0.042233821004629135, "sampling/importance_sampling_ratio/max": 1.8066370487213135, "sampling/importance_sampling_ratio/mean": 0.9998170137405396, "sampling/importance_sampling_ratio/min": 0.28208810091018677, "sampling/sampling_logp_difference/max": 1.265535831451416, "sampling/sampling_logp_difference/mean": 0.016247009858489037, "step": 2010 }, { "clip_ratio/high_max": 0.028711115941405296, "clip_ratio/high_mean": 0.028711115941405296, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/region_mean": 0.032089494401589036, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 74.0, "completions/mean_terminated_length": 74.0, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.35354609973728657, "epoch": 0.08077278386954252, "frac_reward_zero_std": 0.0, "grad_norm": 6.366806983947754, "learning_rate": 3.90909090909091e-06, "loss": 0.0002, "num_tokens": 4528044.0, "reward": 0.9870051145553589, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9870051145553589, "reward_meter_std": 0.02932342328131199, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.029323413968086243, "reward_total_composite_mean": 0.9870051145553589, "reward_total_composite_std": 0.02932342328131199, "reward_total_mean": 0.9870051145553589, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9870051145553589, "rewards/meter/std": 0.02932342328131199, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9870051145553589, "rewards/total_composite/std": 0.02932342328131199, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0154587030410767, "sampling/importance_sampling_ratio/min": 0.2677222192287445, "sampling/sampling_logp_difference/max": 1.317805290222168, "sampling/sampling_logp_difference/mean": 0.04717674106359482, "step": 2011 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 33.0, "completions/mean_terminated_length": 33.0, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.025131846428848803, "epoch": 0.08081294935132748, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.906060606060606e-06, "loss": 0.0, "num_tokens": 4529676.0, "reward": 0.9990787506103516, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990787506103516, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9990787506103516, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9990787506103516, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990787506103516, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990787506103516, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.086820363998413, "sampling/importance_sampling_ratio/mean": 1.0008872747421265, "sampling/importance_sampling_ratio/min": 0.8097277879714966, "sampling/sampling_logp_difference/max": 0.21105718612670898, "sampling/sampling_logp_difference/mean": 0.00452659884467721, "step": 2012 }, { "clip_ratio/high_max": 0.005494505632668734, "clip_ratio/high_mean": 0.005494505632668734, "clip_ratio/low_mean": 0.009800249827094376, "clip_ratio/low_min": 0.009800249827094376, "clip_ratio/region_mean": 0.01529475545976311, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 89.25, "completions/mean_terminated_length": 89.25, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.09196147602051497, "epoch": 0.08085311483311243, "frac_reward_zero_std": 0.0, "grad_norm": 1.6898747682571411, "learning_rate": 3.9030303030303035e-06, "loss": 0.0012, "num_tokens": 4531806.0, "reward": 0.995819091796875, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.995819091796875, "reward_meter_std": 0.0001791357935871929, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00017913819465320557, "reward_total_composite_mean": 0.995819091796875, "reward_total_composite_std": 0.0001791357935871929, "reward_total_mean": 0.995819091796875, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.995819091796875, "rewards/meter/std": 0.0001791357935871929, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.995819091796875, "rewards/total_composite/std": 0.0001791357935871929, "sampling/importance_sampling_ratio/max": 1.4373438358306885, "sampling/importance_sampling_ratio/mean": 1.0000059604644775, "sampling/importance_sampling_ratio/min": 0.19927427172660828, "sampling/sampling_logp_difference/max": 1.6130731105804443, "sampling/sampling_logp_difference/mean": 0.015769556164741516, "step": 2013 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.015853506047278643, "epoch": 0.08089328031489738, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.900000000000001e-06, "loss": 0.0, "num_tokens": 4533534.0, "reward": 0.9970200061798096, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970200061798096, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9970200061798096, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9970200061798096, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970200061798096, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970200061798096, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0803717374801636, "sampling/importance_sampling_ratio/mean": 1.0017708539962769, "sampling/importance_sampling_ratio/min": 0.9277377724647522, "sampling/sampling_logp_difference/max": 0.07730516046285629, "sampling/sampling_logp_difference/mean": 0.002094635274261236, "step": 2014 }, { "clip_ratio/high_max": 0.015477340901270509, "clip_ratio/high_mean": 0.015477340901270509, "clip_ratio/low_mean": 0.0026041667442768812, "clip_ratio/low_min": 0.0026041667442768812, "clip_ratio/region_mean": 0.01808150764554739, "completions/clipped_ratio": 0.0, "completions/max_length": 97.0, "completions/max_terminated_length": 97.0, "completions/mean_length": 96.625, "completions/mean_terminated_length": 96.625, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.11758578661829233, "epoch": 0.08093344579668234, "frac_reward_zero_std": 0.0, "grad_norm": 1.428508996963501, "learning_rate": 3.896969696969697e-06, "loss": 0.0009, "num_tokens": 4535555.0, "reward": 0.9975524544715881, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975524544715881, "reward_meter_std": 0.0001665082381805405, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00016649799363221973, "reward_total_composite_mean": 0.9975524544715881, "reward_total_composite_std": 0.0001665082381805405, "reward_total_mean": 0.9975524544715881, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975524544715881, "rewards/meter/std": 0.0001665082381805405, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975524544715881, "rewards/total_composite/std": 0.0001665082381805405, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0018181800842285, "sampling/importance_sampling_ratio/min": 0.47126853466033936, "sampling/sampling_logp_difference/max": 0.7523272037506104, "sampling/sampling_logp_difference/mean": 0.018747342750430107, "step": 2015 }, { "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.011363636702299118, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.625, "completions/mean_terminated_length": 32.625, "completions/min_length": 30.0, "completions/min_terminated_length": 30.0, "entropy": 0.028148290934041142, "epoch": 0.08097361127846729, "frac_reward_zero_std": 0.0, "grad_norm": 5.452454566955566, "learning_rate": 3.8939393939393944e-06, "loss": -0.035, "num_tokens": 4536992.0, "reward": 0.9971754550933838, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971754550933838, "reward_meter_std": 0.004386701621115208, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0043867104686796665, "reward_total_composite_mean": 0.9971754550933838, "reward_total_composite_std": 0.004386701621115208, "reward_total_mean": 0.9971754550933838, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971754550933838, "rewards/meter/std": 0.004386701621115208, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971754550933838, "rewards/total_composite/std": 0.004386701621115208, "sampling/importance_sampling_ratio/max": 1.2878613471984863, "sampling/importance_sampling_ratio/mean": 0.9991469979286194, "sampling/importance_sampling_ratio/min": 0.3577660620212555, "sampling/sampling_logp_difference/max": 1.0278760194778442, "sampling/sampling_logp_difference/mean": 0.013498242013156414, "step": 2016 }, { "clip_ratio/high_max": 0.007604895276017487, "clip_ratio/high_mean": 0.007604895276017487, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/region_mean": 0.013286713627167046, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.25, "completions/mean_terminated_length": 66.25, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.06328104855492711, "epoch": 0.08101377676025225, "frac_reward_zero_std": 0.0, "grad_norm": 4.091221332550049, "learning_rate": 3.890909090909092e-06, "loss": 0.0048, "num_tokens": 4538962.0, "reward": 0.9984210729598999, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984210729598999, "reward_meter_std": 0.0010966768022626638, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010966742411255836, "reward_total_composite_mean": 0.9984210729598999, "reward_total_composite_std": 0.0010966768022626638, "reward_total_mean": 0.9984210729598999, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984210729598999, "rewards/meter/std": 0.0010966768022626638, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984210729598999, "rewards/total_composite/std": 0.0010966768022626638, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9979748129844666, "sampling/importance_sampling_ratio/min": 0.18793681263923645, "sampling/sampling_logp_difference/max": 1.67164945602417, "sampling/sampling_logp_difference/mean": 0.02255828306078911, "step": 2017 }, { "clip_ratio/high_max": 0.0331978602334857, "clip_ratio/high_mean": 0.0331978602334857, "clip_ratio/low_mean": 0.011172067141160369, "clip_ratio/low_min": 0.011172067141160369, "clip_ratio/region_mean": 0.04436992737464607, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 62.75, "completions/mean_terminated_length": 62.75, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.28926723077893257, "epoch": 0.0810539422420372, "frac_reward_zero_std": 0.0, "grad_norm": 9.685022354125977, "learning_rate": 3.887878787878788e-06, "loss": 0.0544, "num_tokens": 4540736.0, "reward": 0.5434250831604004, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7036042213439941, "reward_meter_std": 0.2752836048603058, "reward_repeat_penalty_mean": 0.800000011920929, "reward_repeat_penalty_std": 0.1511857807636261, "reward_std": 0.20307794213294983, "reward_total_composite_mean": 0.5434250831604004, "reward_total_composite_std": 0.20307792723178864, "reward_total_mean": 0.5434250831604004, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7036042213439941, "rewards/meter/std": 0.2752836048603058, "rewards/repeat_penalty/mean": 0.800000011920929, "rewards/repeat_penalty/std": 0.1511857807636261, "rewards/total_composite/mean": 0.5434250831604004, "rewards/total_composite/std": 0.20307792723178864, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0063507556915283, "sampling/importance_sampling_ratio/min": 0.1436930149793625, "sampling/sampling_logp_difference/max": 1.9400761127471924, "sampling/sampling_logp_difference/mean": 0.0515943206846714, "step": 2018 }, { "clip_ratio/high_max": 0.013733877101913095, "clip_ratio/high_mean": 0.013733877101913095, "clip_ratio/low_mean": 0.014337376691401005, "clip_ratio/low_min": 0.014337376691401005, "clip_ratio/region_mean": 0.0280712537933141, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 106.5, "completions/mean_terminated_length": 106.5, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.3407053891569376, "epoch": 0.08109410772382215, "frac_reward_zero_std": 0.0, "grad_norm": 3.233832359313965, "learning_rate": 3.884848484848485e-06, "loss": -0.0182, "num_tokens": 4542988.0, "reward": 0.9943101406097412, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943101406097412, "reward_meter_std": 0.0025435402058064938, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0025435429997742176, "reward_total_composite_mean": 0.9943101406097412, "reward_total_composite_std": 0.0025435402058064938, "reward_total_mean": 0.9943101406097412, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943101406097412, "rewards/meter/std": 0.0025435402058064938, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943101406097412, "rewards/total_composite/std": 0.0025435402058064938, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010913610458374, "sampling/importance_sampling_ratio/min": 0.22902902960777283, "sampling/sampling_logp_difference/max": 1.4739065170288086, "sampling/sampling_logp_difference/mean": 0.03694016858935356, "step": 2019 }, { "clip_ratio/high_max": 0.01079247216694057, "clip_ratio/high_mean": 0.01079247216694057, "clip_ratio/low_mean": 0.0007225433364510536, "clip_ratio/low_min": 0.0007225433364510536, "clip_ratio/region_mean": 0.011515015503391623, "completions/clipped_ratio": 0.0, "completions/max_length": 174.0, "completions/max_terminated_length": 174.0, "completions/mean_length": 173.75, "completions/mean_terminated_length": 173.75, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.07485386310145259, "epoch": 0.08113427320560711, "frac_reward_zero_std": 0.0, "grad_norm": 2.484889507293701, "learning_rate": 3.881818181818182e-06, "loss": 0.0002, "num_tokens": 4545802.0, "reward": 0.6928579211235046, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9751757383346558, "reward_meter_std": 0.06318987905979156, "reward_repeat_penalty_mean": 0.7124999761581421, "reward_repeat_penalty_std": 0.0353553481400013, "reward_std": 0.015284637920558453, "reward_total_composite_mean": 0.6928579211235046, "reward_total_composite_std": 0.015284620225429535, "reward_total_mean": 0.6928579211235046, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9751757383346558, "rewards/meter/std": 0.06318987905979156, "rewards/repeat_penalty/mean": 0.7124999761581421, "rewards/repeat_penalty/std": 0.0353553481400013, "rewards/total_composite/mean": 0.6928579211235046, "rewards/total_composite/std": 0.015284620225429535, "sampling/importance_sampling_ratio/max": 1.7495958805084229, "sampling/importance_sampling_ratio/mean": 1.0012462139129639, "sampling/importance_sampling_ratio/min": 0.2607121467590332, "sampling/sampling_logp_difference/max": 1.3443384170532227, "sampling/sampling_logp_difference/mean": 0.013370201922953129, "step": 2020 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 137.0, "completions/mean_terminated_length": 137.0, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.011890202295035124, "epoch": 0.08117443868739206, "frac_reward_zero_std": 0.0, "grad_norm": 1.3713654279708862, "learning_rate": 3.878787878787879e-06, "loss": 0.0001, "num_tokens": 4548138.0, "reward": 0.7756081223487854, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.904876172542572, "reward_meter_std": 0.0019419160671532154, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0016644754214212298, "reward_total_composite_mean": 0.7756081223487854, "reward_total_composite_std": 0.0016644843854010105, "reward_total_mean": 0.7756081223487854, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.904876172542572, "rewards/meter/std": 0.0019419160671532154, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7756081223487854, "rewards/total_composite/std": 0.0016644843854010105, "sampling/importance_sampling_ratio/max": 1.263245940208435, "sampling/importance_sampling_ratio/mean": 1.0013093948364258, "sampling/importance_sampling_ratio/min": 0.9551554918289185, "sampling/sampling_logp_difference/max": 0.2336844801902771, "sampling/sampling_logp_difference/mean": 0.001481961109675467, "step": 2021 }, { "clip_ratio/high_max": 0.014087301678955555, "clip_ratio/high_mean": 0.014087301678955555, "clip_ratio/low_mean": 0.011029412038624287, "clip_ratio/low_min": 0.011029412038624287, "clip_ratio/region_mean": 0.02511671371757984, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.625, "completions/mean_terminated_length": 34.625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.07363486429676414, "epoch": 0.08121460416917702, "frac_reward_zero_std": 0.0, "grad_norm": 6.894834518432617, "learning_rate": 3.875757575757576e-06, "loss": -0.0341, "num_tokens": 4549759.0, "reward": 0.9927211403846741, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9927211403846741, "reward_meter_std": 0.009707700461149216, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009707717224955559, "reward_total_composite_mean": 0.9927211403846741, "reward_total_composite_std": 0.009707700461149216, "reward_total_mean": 0.9927211403846741, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9927211403846741, "rewards/meter/std": 0.009707700461149216, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9927211403846741, "rewards/total_composite/std": 0.009707700461149216, "sampling/importance_sampling_ratio/max": 1.2991583347320557, "sampling/importance_sampling_ratio/mean": 0.9955629110336304, "sampling/importance_sampling_ratio/min": 0.2899340093135834, "sampling/sampling_logp_difference/max": 1.2381019592285156, "sampling/sampling_logp_difference/mean": 0.02147146500647068, "step": 2022 }, { "clip_ratio/high_max": 0.006896879756823182, "clip_ratio/high_mean": 0.006896879756823182, "clip_ratio/low_mean": 0.005184551002457738, "clip_ratio/low_min": 0.005184551002457738, "clip_ratio/region_mean": 0.01208143075928092, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.5, "completions/mean_terminated_length": 72.5, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.24161814339458942, "epoch": 0.08125476965096197, "frac_reward_zero_std": 0.0, "grad_norm": 2.7678604125976562, "learning_rate": 3.872727272727273e-06, "loss": 0.0061, "num_tokens": 4551755.0, "reward": 0.9986792802810669, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986792802810669, "reward_meter_std": 0.000637872377410531, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006378723192028701, "reward_total_composite_mean": 0.9986792802810669, "reward_total_composite_std": 0.000637872377410531, "reward_total_mean": 0.9986792802810669, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986792802810669, "rewards/meter/std": 0.000637872377410531, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986792802810669, "rewards/total_composite/std": 0.000637872377410531, "sampling/importance_sampling_ratio/max": 1.7916842699050903, "sampling/importance_sampling_ratio/mean": 1.0089229345321655, "sampling/importance_sampling_ratio/min": 0.4119720160961151, "sampling/sampling_logp_difference/max": 0.8867998123168945, "sampling/sampling_logp_difference/mean": 0.03179734945297241, "step": 2023 }, { "clip_ratio/high_max": 0.006140776677057147, "clip_ratio/high_mean": 0.006140776677057147, "clip_ratio/low_mean": 0.0024038462433964014, "clip_ratio/low_min": 0.0024038462433964014, "clip_ratio/region_mean": 0.008544622920453548, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 102.75, "completions/mean_terminated_length": 102.75, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.05600249394774437, "epoch": 0.08129493513274692, "frac_reward_zero_std": 0.0, "grad_norm": 1.6175122261047363, "learning_rate": 3.86969696969697e-06, "loss": 0.0087, "num_tokens": 4553953.0, "reward": 0.9968816041946411, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9968816041946411, "reward_meter_std": 0.0006491380045190454, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006491419626399875, "reward_total_composite_mean": 0.9968816041946411, "reward_total_composite_std": 0.0006491380045190454, "reward_total_mean": 0.9968816041946411, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9968816041946411, "rewards/meter/std": 0.0006491380045190454, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9968816041946411, "rewards/total_composite/std": 0.0006491380045190454, "sampling/importance_sampling_ratio/max": 1.40573251247406, "sampling/importance_sampling_ratio/mean": 0.9999393820762634, "sampling/importance_sampling_ratio/min": 0.36837852001190186, "sampling/sampling_logp_difference/max": 0.9986443519592285, "sampling/sampling_logp_difference/mean": 0.010959350503981113, "step": 2024 }, { "clip_ratio/high_max": 0.025219141854904592, "clip_ratio/high_mean": 0.025219141854904592, "clip_ratio/low_mean": 0.017962860874831676, "clip_ratio/low_min": 0.017962860874831676, "clip_ratio/region_mean": 0.04318200272973627, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 150.375, "completions/mean_terminated_length": 150.375, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.2885005250573158, "epoch": 0.08133510061453188, "frac_reward_zero_std": 0.0, "grad_norm": 3.409259557723999, "learning_rate": 3.866666666666667e-06, "loss": 0.0171, "num_tokens": 4556516.0, "reward": 0.8427009582519531, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.890097975730896, "reward_meter_std": 0.1463252604007721, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.1560104936361313, "reward_total_composite_mean": 0.8427009582519531, "reward_total_composite_std": 0.15601050853729248, "reward_total_mean": 0.8427009582519531, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.890097975730896, "rewards/meter/std": 0.1463252604007721, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.8427009582519531, "rewards/total_composite/std": 0.15601050853729248, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9980630278587341, "sampling/importance_sampling_ratio/min": 0.021800734102725983, "sampling/sampling_logp_difference/max": 3.8258116245269775, "sampling/sampling_logp_difference/mean": 0.05061601847410202, "step": 2025 }, { "clip_ratio/high_max": 0.003759611048735678, "clip_ratio/high_mean": 0.003759611048735678, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/region_mean": 0.007490954245440662, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.038185699842870235, "epoch": 0.08137526609631683, "frac_reward_zero_std": 0.0, "grad_norm": 2.6563684940338135, "learning_rate": 3.863636363636364e-06, "loss": 0.0053, "num_tokens": 4558243.0, "reward": 0.9982448220252991, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982448220252991, "reward_meter_std": 0.0003502867475617677, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00035030534490942955, "reward_total_composite_mean": 0.9982448220252991, "reward_total_composite_std": 0.0003502867475617677, "reward_total_mean": 0.9982448220252991, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982448220252991, "rewards/meter/std": 0.0003502867475617677, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982448220252991, "rewards/total_composite/std": 0.0003502867475617677, "sampling/importance_sampling_ratio/max": 1.3493366241455078, "sampling/importance_sampling_ratio/mean": 0.9976450800895691, "sampling/importance_sampling_ratio/min": 0.4521390497684479, "sampling/sampling_logp_difference/max": 0.7937655448913574, "sampling/sampling_logp_difference/mean": 0.01347498781979084, "step": 2026 }, { "clip_ratio/high_max": 0.0037086496595293283, "clip_ratio/high_mean": 0.0037086496595293283, "clip_ratio/low_mean": 0.005135256564244628, "clip_ratio/low_min": 0.005135256564244628, "clip_ratio/region_mean": 0.008843906223773956, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 340.0, "completions/mean_terminated_length": 340.0, "completions/min_length": 333.0, "completions/min_terminated_length": 333.0, "entropy": 0.06602455815300345, "epoch": 0.08141543157810179, "frac_reward_zero_std": 0.0, "grad_norm": 2.305088758468628, "learning_rate": 3.860606060606061e-06, "loss": 0.0071, "num_tokens": 4562915.0, "reward": 0.5113018751144409, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.701923131942749, "reward_count_adherence_std": 0.027196412906050682, "reward_meter_mean": 0.9870275259017944, "reward_meter_std": 0.0033544274047017097, "reward_repeat_penalty_mean": 0.7379385828971863, "reward_repeat_penalty_std": 0.02510136552155018, "reward_std": 0.027315424755215645, "reward_total_composite_mean": 0.5113018751144409, "reward_total_composite_std": 0.02731543779373169, "reward_total_mean": 0.5113018751144409, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.701923131942749, "rewards/count_adherence/std": 0.027196412906050682, "rewards/meter/mean": 0.9870275259017944, "rewards/meter/std": 0.0033544274047017097, "rewards/repeat_penalty/mean": 0.7379385828971863, "rewards/repeat_penalty/std": 0.02510136552155018, "rewards/total_composite/mean": 0.5113018751144409, "rewards/total_composite/std": 0.02731543779373169, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0029667615890503, "sampling/importance_sampling_ratio/min": 0.17998412251472473, "sampling/sampling_logp_difference/max": 1.7148866653442383, "sampling/sampling_logp_difference/mean": 0.013775442726910114, "step": 2027 }, { "clip_ratio/high_max": 0.007299659075215459, "clip_ratio/high_mean": 0.007299659075215459, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/region_mean": 0.0090853733709082, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.06549654342234135, "epoch": 0.08145559705988674, "frac_reward_zero_std": 0.0, "grad_norm": 4.434993743896484, "learning_rate": 3.857575757575758e-06, "loss": 0.0109, "num_tokens": 4564619.0, "reward": 0.9325088262557983, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9325088262557983, "reward_meter_std": 0.02886304259300232, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02886303886771202, "reward_total_composite_mean": 0.9325088262557983, "reward_total_composite_std": 0.02886304259300232, "reward_total_mean": 0.9325088262557983, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9325088262557983, "rewards/meter/std": 0.02886304259300232, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9325088262557983, "rewards/total_composite/std": 0.02886304259300232, "sampling/importance_sampling_ratio/max": 1.2660948038101196, "sampling/importance_sampling_ratio/mean": 0.9968265295028687, "sampling/importance_sampling_ratio/min": 0.36934226751327515, "sampling/sampling_logp_difference/max": 0.9960315227508545, "sampling/sampling_logp_difference/mean": 0.018320947885513306, "step": 2028 }, { "clip_ratio/high_max": 0.0012135922443121672, "clip_ratio/high_mean": 0.0012135922443121672, "clip_ratio/low_mean": 0.0036526747280731797, "clip_ratio/low_min": 0.0036526747280731797, "clip_ratio/region_mean": 0.004866266972385347, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 102.75, "completions/mean_terminated_length": 102.75, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.04349301569163799, "epoch": 0.08149576254167169, "frac_reward_zero_std": 0.0, "grad_norm": 0.44382914900779724, "learning_rate": 3.8545454545454545e-06, "loss": -0.0021, "num_tokens": 4566681.0, "reward": 0.9243794083595276, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9243794083595276, "reward_meter_std": 0.014914168044924736, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01491414662450552, "reward_total_composite_mean": 0.9243794083595276, "reward_total_composite_std": 0.014914168044924736, "reward_total_mean": 0.9243794083595276, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9243794083595276, "rewards/meter/std": 0.014914168044924736, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9243794083595276, "rewards/total_composite/std": 0.014914168044924736, "sampling/importance_sampling_ratio/max": 1.119174838066101, "sampling/importance_sampling_ratio/mean": 1.002354621887207, "sampling/importance_sampling_ratio/min": 0.733018696308136, "sampling/sampling_logp_difference/max": 0.31058406829833984, "sampling/sampling_logp_difference/mean": 0.004690106958150864, "step": 2029 }, { "clip_ratio/high_max": 0.019332326017320156, "clip_ratio/high_mean": 0.019332326017320156, "clip_ratio/low_mean": 0.013769977260380983, "clip_ratio/low_min": 0.013769977260380983, "clip_ratio/region_mean": 0.03310230327770114, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 47.25, "completions/mean_terminated_length": 47.25, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.3015878004953265, "epoch": 0.08153592802345665, "frac_reward_zero_std": 0.0, "grad_norm": 10.0611572265625, "learning_rate": 3.851515151515152e-06, "loss": 0.3073, "num_tokens": 4568259.0, "reward": 0.7492817640304565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_meter_mean": 0.9984946846961975, "reward_meter_std": 0.0014516202500090003, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.46246692538261414, "reward_total_composite_mean": 0.7492817640304565, "reward_total_composite_std": 0.46246692538261414, "reward_total_mean": 0.7492817640304565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/meter/mean": 0.9984946846961975, "rewards/meter/std": 0.0014516202500090003, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7492817640304565, "rewards/total_composite/std": 0.46246692538261414, "sampling/importance_sampling_ratio/max": 1.675675392150879, "sampling/importance_sampling_ratio/mean": 1.003790259361267, "sampling/importance_sampling_ratio/min": 0.12250065058469772, "sampling/sampling_logp_difference/max": 2.0996389389038086, "sampling/sampling_logp_difference/mean": 0.057400476187467575, "step": 2030 }, { "clip_ratio/high_max": 0.004751632455736399, "clip_ratio/high_mean": 0.004751632455736399, "clip_ratio/low_mean": 0.011571898590773344, "clip_ratio/low_min": 0.011571898590773344, "clip_ratio/region_mean": 0.016323531046509743, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 129.375, "completions/mean_terminated_length": 129.375, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.06218431191518903, "epoch": 0.0815760935052416, "frac_reward_zero_std": 0.0, "grad_norm": 3.1920166015625, "learning_rate": 3.848484848484848e-06, "loss": -0.004, "num_tokens": 4570662.0, "reward": 0.8908217549324036, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977312684059143, "reward_meter_std": 0.00020651114755310118, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06581789255142212, "reward_total_composite_mean": 0.8908217549324036, "reward_total_composite_std": 0.06581787765026093, "reward_total_mean": 0.8908217549324036, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977312684059143, "rewards/meter/std": 0.00020651114755310118, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8908217549324036, "rewards/total_composite/std": 0.06581787765026093, "sampling/importance_sampling_ratio/max": 1.6425533294677734, "sampling/importance_sampling_ratio/mean": 1.0028420686721802, "sampling/importance_sampling_ratio/min": 0.23743832111358643, "sampling/sampling_logp_difference/max": 1.437847375869751, "sampling/sampling_logp_difference/mean": 0.01242055743932724, "step": 2031 }, { "clip_ratio/high_max": 0.009469697251915932, "clip_ratio/high_mean": 0.009469697251915932, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.013257576152682304, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.06596994632855058, "epoch": 0.08161625898702655, "frac_reward_zero_std": 0.0, "grad_norm": 1.8542088270187378, "learning_rate": 3.8454545454545454e-06, "loss": 0.0009, "num_tokens": 4572734.0, "reward": 0.9978283643722534, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978283643722534, "reward_meter_std": 0.00016764015890657902, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00016763212624937296, "reward_total_composite_mean": 0.9978283643722534, "reward_total_composite_std": 0.00016764015890657902, "reward_total_mean": 0.9978283643722534, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978283643722534, "rewards/meter/std": 0.00016764015890657902, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978283643722534, "rewards/total_composite/std": 0.00016764015890657902, "sampling/importance_sampling_ratio/max": 1.2213056087493896, "sampling/importance_sampling_ratio/mean": 1.0011532306671143, "sampling/importance_sampling_ratio/min": 0.6987596154212952, "sampling/sampling_logp_difference/max": 0.35844850540161133, "sampling/sampling_logp_difference/mean": 0.011215658858418465, "step": 2032 }, { "clip_ratio/high_max": 0.023466870421543717, "clip_ratio/high_mean": 0.023466870421543717, "clip_ratio/low_mean": 0.010510046035051346, "clip_ratio/low_min": 0.010510046035051346, "clip_ratio/region_mean": 0.03397691645659506, "completions/clipped_ratio": 0.0, "completions/max_length": 163.0, "completions/max_terminated_length": 163.0, "completions/mean_length": 148.375, "completions/mean_terminated_length": 148.375, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "entropy": 0.24012545123696327, "epoch": 0.08165642446881151, "frac_reward_zero_std": 0.0, "grad_norm": 3.462294578552246, "learning_rate": 3.842424242424243e-06, "loss": 0.0323, "num_tokens": 4575273.0, "reward": 0.8339765071868896, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.91922527551651, "reward_meter_std": 0.09434106945991516, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.08177479356527328, "reward_total_composite_mean": 0.8339765071868896, "reward_total_composite_std": 0.08177480101585388, "reward_total_mean": 0.8339765071868896, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.91922527551651, "rewards/meter/std": 0.09434106945991516, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.8339765071868896, "rewards/total_composite/std": 0.08177480101585388, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0021495819091797, "sampling/importance_sampling_ratio/min": 0.18345968425273895, "sampling/sampling_logp_difference/max": 1.6957603693008423, "sampling/sampling_logp_difference/mean": 0.04417654499411583, "step": 2033 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.024745611241087317, "epoch": 0.08169658995059646, "frac_reward_zero_std": 0.0, "grad_norm": 0.040686190128326416, "learning_rate": 3.839393939393939e-06, "loss": 0.0002, "num_tokens": 4576945.0, "reward": 0.997020959854126, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997020959854126, "reward_meter_std": 2.6552993404038716e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.661313374119345e-06, "reward_total_composite_mean": 0.997020959854126, "reward_total_composite_std": 2.6552993404038716e-06, "reward_total_mean": 0.997020959854126, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997020959854126, "rewards/meter/std": 2.6552993404038716e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997020959854126, "rewards/total_composite/std": 2.6552993404038716e-06, "sampling/importance_sampling_ratio/max": 1.0738173723220825, "sampling/importance_sampling_ratio/mean": 1.0010058879852295, "sampling/importance_sampling_ratio/min": 0.6608229279518127, "sampling/sampling_logp_difference/max": 0.41426944732666016, "sampling/sampling_logp_difference/mean": 0.0038556025829166174, "step": 2034 }, { "clip_ratio/high_max": 0.0169863011687994, "clip_ratio/high_mean": 0.0169863011687994, "clip_ratio/low_mean": 0.008355855825357139, "clip_ratio/low_min": 0.008355855825357139, "clip_ratio/region_mean": 0.02534215699415654, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 74.25, "completions/mean_terminated_length": 74.25, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.2636965289711952, "epoch": 0.08173675543238142, "frac_reward_zero_std": 0.0, "grad_norm": 5.467068195343018, "learning_rate": 3.836363636363636e-06, "loss": 0.0137, "num_tokens": 4578843.0, "reward": 0.7550032138824463, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8532536625862122, "reward_meter_std": 0.21336820721626282, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3712865710258484, "reward_total_composite_mean": 0.7550032138824463, "reward_total_composite_std": 0.3712865710258484, "reward_total_mean": 0.7550032138824463, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8532536625862122, "rewards/meter/std": 0.21336820721626282, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7550032138824463, "rewards/total_composite/std": 0.3712865710258484, "sampling/importance_sampling_ratio/max": 1.8733973503112793, "sampling/importance_sampling_ratio/mean": 1.005318522453308, "sampling/importance_sampling_ratio/min": 0.1933393031358719, "sampling/sampling_logp_difference/max": 1.6433086395263672, "sampling/sampling_logp_difference/mean": 0.04074222594499588, "step": 2035 }, { "clip_ratio/high_max": 0.016197497956454754, "clip_ratio/high_mean": 0.016197497956454754, "clip_ratio/low_mean": 0.0022935778833925724, "clip_ratio/low_min": 0.0022935778833925724, "clip_ratio/region_mean": 0.018491075839847326, "completions/clipped_ratio": 0.0, "completions/max_length": 110.0, "completions/max_terminated_length": 110.0, "completions/mean_length": 108.125, "completions/mean_terminated_length": 108.125, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.21733380481600761, "epoch": 0.08177692091416637, "frac_reward_zero_std": 0.0, "grad_norm": 1.754205584526062, "learning_rate": 3.833333333333334e-06, "loss": 0.003, "num_tokens": 4581108.0, "reward": 0.9737250804901123, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986991882324219, "reward_meter_std": 0.0005537345423363149, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07053209841251373, "reward_total_composite_mean": 0.9737250804901123, "reward_total_composite_std": 0.07053209096193314, "reward_total_mean": 0.9737250804901123, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986991882324219, "rewards/meter/std": 0.0005537345423363149, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9737250804901123, "rewards/total_composite/std": 0.07053209096193314, "sampling/importance_sampling_ratio/max": 1.7996537685394287, "sampling/importance_sampling_ratio/mean": 1.00703763961792, "sampling/importance_sampling_ratio/min": 0.37958869338035583, "sampling/sampling_logp_difference/max": 0.9686670303344727, "sampling/sampling_logp_difference/mean": 0.02059764787554741, "step": 2036 }, { "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.005681818351149559, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.040048055816441774, "epoch": 0.08181708639595132, "frac_reward_zero_std": 0.0, "grad_norm": 0.042626433074474335, "learning_rate": 3.830303030303031e-06, "loss": 0.0, "num_tokens": 4582996.0, "reward": 0.9979811906814575, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979811906814575, "reward_meter_std": 3.9898877730593085e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.0090494621836115e-06, "reward_total_composite_mean": 0.9979811906814575, "reward_total_composite_std": 3.9898877730593085e-06, "reward_total_mean": 0.9979811906814575, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979811906814575, "rewards/meter/std": 3.9898877730593085e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979811906814575, "rewards/total_composite/std": 3.9898877730593085e-06, "sampling/importance_sampling_ratio/max": 1.3170571327209473, "sampling/importance_sampling_ratio/mean": 1.0025206804275513, "sampling/importance_sampling_ratio/min": 0.6638693809509277, "sampling/sampling_logp_difference/max": 0.40966981649398804, "sampling/sampling_logp_difference/mean": 0.00694712670519948, "step": 2037 }, { "clip_ratio/high_max": 0.0038659792626276612, "clip_ratio/high_mean": 0.0038659792626276612, "clip_ratio/low_mean": 0.0012886597542092204, "clip_ratio/low_min": 0.0012886597542092204, "clip_ratio/region_mean": 0.005154639016836882, "completions/clipped_ratio": 0.0, "completions/max_length": 97.0, "completions/max_terminated_length": 97.0, "completions/mean_length": 97.0, "completions/mean_terminated_length": 97.0, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.025324456859380007, "epoch": 0.08185725187773628, "frac_reward_zero_std": 0.0, "grad_norm": 0.05830654874444008, "learning_rate": 3.827272727272728e-06, "loss": 0.0002, "num_tokens": 4585172.0, "reward": 0.9979192614555359, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979192614555359, "reward_meter_std": 5.3105509323359e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.296794824971585e-06, "reward_total_composite_mean": 0.9979192614555359, "reward_total_composite_std": 5.3105509323359e-06, "reward_total_mean": 0.9979192614555359, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979192614555359, "rewards/meter/std": 5.3105509323359e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979192614555359, "rewards/total_composite/std": 5.3105509323359e-06, "sampling/importance_sampling_ratio/max": 1.242266297340393, "sampling/importance_sampling_ratio/mean": 1.0008537769317627, "sampling/importance_sampling_ratio/min": 0.6097805500030518, "sampling/sampling_logp_difference/max": 0.49465620517730713, "sampling/sampling_logp_difference/mean": 0.0037312970962375402, "step": 2038 }, { "clip_ratio/high_max": 0.02182656608056277, "clip_ratio/high_mean": 0.02182656608056277, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/region_mean": 0.026212531025521457, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 110.25, "completions/mean_terminated_length": 110.25, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.32149537093937397, "epoch": 0.08189741735952123, "frac_reward_zero_std": 0.0, "grad_norm": 10.090517044067383, "learning_rate": 3.8242424242424245e-06, "loss": 0.0239, "num_tokens": 4587310.0, "reward": 0.9696316123008728, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942349195480347, "reward_meter_std": 0.0053013949654996395, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07374817132949829, "reward_total_composite_mean": 0.9696316123008728, "reward_total_composite_std": 0.07374817132949829, "reward_total_mean": 0.9696316123008728, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942349195480347, "rewards/meter/std": 0.0053013949654996395, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9696316123008728, "rewards/total_composite/std": 0.07374817132949829, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0104196071624756, "sampling/importance_sampling_ratio/min": 0.3576737940311432, "sampling/sampling_logp_difference/max": 1.0281339883804321, "sampling/sampling_logp_difference/mean": 0.038861531764268875, "step": 2039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.009250407747458667, "epoch": 0.08193758284130619, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.821212121212122e-06, "loss": 0.0, "num_tokens": 4589110.0, "reward": 0.9990598559379578, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9990598559379578, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0154409408569336, "sampling/importance_sampling_ratio/mean": 1.0006673336029053, "sampling/importance_sampling_ratio/min": 0.9778358340263367, "sampling/sampling_logp_difference/max": 0.022413522005081177, "sampling/sampling_logp_difference/mean": 0.0009253994794562459, "step": 2040 }, { "clip_ratio/high_max": 0.009615384740754962, "clip_ratio/high_mean": 0.009615384740754962, "clip_ratio/low_mean": 0.006330128293484449, "clip_ratio/low_min": 0.006330128293484449, "clip_ratio/region_mean": 0.01594551303423941, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 39.0, "completions/mean_terminated_length": 39.0, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.21922382153570652, "epoch": 0.08197774832309114, "frac_reward_zero_std": 0.0, "grad_norm": 4.351480007171631, "learning_rate": 3.818181818181819e-06, "loss": -0.0008, "num_tokens": 4590654.0, "reward": 0.9944432973861694, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9944432973861694, "reward_meter_std": 0.0023516512010246515, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0023516479413956404, "reward_total_composite_mean": 0.9944432973861694, "reward_total_composite_std": 0.0023516512010246515, "reward_total_mean": 0.9944432973861694, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9944432973861694, "rewards/meter/std": 0.0023516512010246515, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944432973861694, "rewards/total_composite/std": 0.0023516512010246515, "sampling/importance_sampling_ratio/max": 1.7522445917129517, "sampling/importance_sampling_ratio/mean": 1.0104680061340332, "sampling/importance_sampling_ratio/min": 0.5875173807144165, "sampling/sampling_logp_difference/max": 0.5608975887298584, "sampling/sampling_logp_difference/mean": 0.022262930870056152, "step": 2041 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/region_mean": 0.0021551724057644606, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.023268206976354122, "epoch": 0.0820179138048761, "frac_reward_zero_std": 0.0, "grad_norm": 0.49870744347572327, "learning_rate": 3.8151515151515155e-06, "loss": 0.0002, "num_tokens": 4592318.0, "reward": 0.9948880672454834, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948880672454834, "reward_meter_std": 3.073411789955571e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.0727183911949396e-05, "reward_total_composite_mean": 0.9948880672454834, "reward_total_composite_std": 3.073411789955571e-05, "reward_total_mean": 0.9948880672454834, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948880672454834, "rewards/meter/std": 3.073411789955571e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948880672454834, "rewards/total_composite/std": 3.073411789955571e-05, "sampling/importance_sampling_ratio/max": 1.819435477256775, "sampling/importance_sampling_ratio/mean": 1.001407265663147, "sampling/importance_sampling_ratio/min": 0.44524553418159485, "sampling/sampling_logp_difference/max": 0.8091294765472412, "sampling/sampling_logp_difference/mean": 0.004786766599863768, "step": 2042 }, { "clip_ratio/high_max": 0.01007624133490026, "clip_ratio/high_mean": 0.01007624133490026, "clip_ratio/low_mean": 0.0013192612677812576, "clip_ratio/low_min": 0.0013192612677812576, "clip_ratio/region_mean": 0.011395502602681518, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 382.125, "completions/mean_terminated_length": 382.125, "completions/min_length": 379.0, "completions/min_terminated_length": 379.0, "entropy": 0.07445414271205664, "epoch": 0.08205807928666105, "frac_reward_zero_std": 0.0, "grad_norm": 2.7885191440582275, "learning_rate": 3.8121212121212127e-06, "loss": -0.0015, "num_tokens": 4597255.0, "reward": 0.42605897784233093, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6397058963775635, "reward_count_adherence_std": 0.020797256380319595, "reward_meter_mean": 0.9941146373748779, "reward_meter_std": 0.0013542358065024018, "reward_repeat_penalty_mean": 0.6714285612106323, "reward_repeat_penalty_std": 0.06172133609652519, "reward_std": 0.02748279646039009, "reward_total_composite_mean": 0.42605897784233093, "reward_total_composite_std": 0.027482789009809494, "reward_total_mean": 0.42605897784233093, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6397058963775635, "rewards/count_adherence/std": 0.020797256380319595, "rewards/meter/mean": 0.9941146373748779, "rewards/meter/std": 0.0013542358065024018, "rewards/repeat_penalty/mean": 0.6714285612106323, "rewards/repeat_penalty/std": 0.06172133609652519, "rewards/total_composite/mean": 0.42605897784233093, "rewards/total_composite/std": 0.027482789009809494, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0006448030471802, "sampling/importance_sampling_ratio/min": 0.006612143013626337, "sampling/sampling_logp_difference/max": 5.018847465515137, "sampling/sampling_logp_difference/mean": 0.017567651346325874, "step": 2043 }, { "clip_ratio/high_max": 0.00866253802087158, "clip_ratio/high_mean": 0.00866253802087158, "clip_ratio/low_mean": 0.012303744442760944, "clip_ratio/low_min": 0.012303744442760944, "clip_ratio/region_mean": 0.020966282463632524, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 71.75, "completions/mean_terminated_length": 71.75, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.23516237549483776, "epoch": 0.082098244768446, "frac_reward_zero_std": 0.0, "grad_norm": 3.922247886657715, "learning_rate": 3.8090909090909095e-06, "loss": -0.018, "num_tokens": 4599133.0, "reward": 0.9981293082237244, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981293082237244, "reward_meter_std": 0.0010699051199480891, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010698941769078374, "reward_total_composite_mean": 0.9981293082237244, "reward_total_composite_std": 0.0010699051199480891, "reward_total_mean": 0.9981293082237244, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981293082237244, "rewards/meter/std": 0.0010699051199480891, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981293082237244, "rewards/total_composite/std": 0.0010699051199480891, "sampling/importance_sampling_ratio/max": 1.6578000783920288, "sampling/importance_sampling_ratio/mean": 1.008084774017334, "sampling/importance_sampling_ratio/min": 0.2196274846792221, "sampling/sampling_logp_difference/max": 1.515822410583496, "sampling/sampling_logp_difference/mean": 0.03494240716099739, "step": 2044 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.004853463382460177, "clip_ratio/low_min": 0.004853463382460177, "clip_ratio/region_mean": 0.004853463382460177, "completions/clipped_ratio": 0.0, "completions/max_length": 131.0, "completions/max_terminated_length": 131.0, "completions/mean_length": 128.125, "completions/mean_terminated_length": 128.125, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.04036114830523729, "epoch": 0.08213841025023096, "frac_reward_zero_std": 0.0, "grad_norm": 3.759932279586792, "learning_rate": 3.8060606060606064e-06, "loss": -0.0019, "num_tokens": 4601598.0, "reward": 0.8919284343719482, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989614486694336, "reward_meter_std": 9.313080954598263e-05, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06603794544935226, "reward_total_composite_mean": 0.8919284343719482, "reward_total_composite_std": 0.06603793799877167, "reward_total_mean": 0.8919284343719482, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989614486694336, "rewards/meter/std": 9.313080954598263e-05, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8919284343719482, "rewards/total_composite/std": 0.06603793799877167, "sampling/importance_sampling_ratio/max": 1.4929101467132568, "sampling/importance_sampling_ratio/mean": 0.9983217120170593, "sampling/importance_sampling_ratio/min": 0.24597874283790588, "sampling/sampling_logp_difference/max": 1.402510166168213, "sampling/sampling_logp_difference/mean": 0.013363703154027462, "step": 2045 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.007882928592152894, "epoch": 0.08217857573201591, "frac_reward_zero_std": 0.0, "grad_norm": 0.4616295397281647, "learning_rate": 3.803030303030303e-06, "loss": -0.0005, "num_tokens": 4603398.0, "reward": 0.999051570892334, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999051570892334, "reward_meter_std": 2.334937744308263e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.3349353796220385e-05, "reward_total_composite_mean": 0.999051570892334, "reward_total_composite_std": 2.334937744308263e-05, "reward_total_mean": 0.999051570892334, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999051570892334, "rewards/meter/std": 2.334937744308263e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999051570892334, "rewards/total_composite/std": 2.334937744308263e-05, "sampling/importance_sampling_ratio/max": 1.0310380458831787, "sampling/importance_sampling_ratio/mean": 1.0000971555709839, "sampling/importance_sampling_ratio/min": 0.7167844176292419, "sampling/sampling_logp_difference/max": 0.3329801559448242, "sampling/sampling_logp_difference/mean": 0.0014573887456208467, "step": 2046 }, { "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.0037878789007663727, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.03522463422268629, "epoch": 0.08221874121380086, "frac_reward_zero_std": 0.0, "grad_norm": 0.1668306142091751, "learning_rate": 3.8000000000000005e-06, "loss": -0.0001, "num_tokens": 4605142.0, "reward": 0.9979839324951172, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979839324951172, "reward_meter_std": 4.366625034890603e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.369384441815782e-06, "reward_total_composite_mean": 0.9979839324951172, "reward_total_composite_std": 4.366625034890603e-06, "reward_total_mean": 0.9979839324951172, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979839324951172, "rewards/meter/std": 4.366625034890603e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979839324951172, "rewards/total_composite/std": 4.366625034890603e-06, "sampling/importance_sampling_ratio/max": 1.3960132598876953, "sampling/importance_sampling_ratio/mean": 1.0016472339630127, "sampling/importance_sampling_ratio/min": 0.5496423840522766, "sampling/sampling_logp_difference/max": 0.598487377166748, "sampling/sampling_logp_difference/mean": 0.005823140498250723, "step": 2047 }, { "clip_ratio/high_max": 0.018360439455136657, "clip_ratio/high_mean": 0.018360439455136657, "clip_ratio/low_mean": 0.012748151202686131, "clip_ratio/low_min": 0.012748151202686131, "clip_ratio/region_mean": 0.031108590657822788, "completions/clipped_ratio": 0.0, "completions/max_length": 221.0, "completions/max_terminated_length": 221.0, "completions/mean_length": 213.25, "completions/mean_terminated_length": 213.25, "completions/min_length": 187.0, "completions/min_terminated_length": 187.0, "entropy": 0.2369670756161213, "epoch": 0.08225890669558582, "frac_reward_zero_std": 0.0, "grad_norm": 1.707572102546692, "learning_rate": 3.7969696969696973e-06, "loss": -0.0143, "num_tokens": 4608320.0, "reward": 0.8758239150047302, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.9987424612045288, "reward_meter_std": 0.0006651075091212988, "reward_repeat_penalty_mean": 0.8977272510528564, "reward_repeat_penalty_std": 0.09009374678134918, "reward_std": 0.08225142955780029, "reward_total_composite_mean": 0.8758239150047302, "reward_total_composite_std": 0.08225142955780029, "reward_total_mean": 0.8758239150047302, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.9987424612045288, "rewards/meter/std": 0.0006651075091212988, "rewards/repeat_penalty/mean": 0.8977272510528564, "rewards/repeat_penalty/std": 0.09009374678134918, "rewards/total_composite/mean": 0.8758239150047302, "rewards/total_composite/std": 0.08225142955780029, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006170392036438, "sampling/importance_sampling_ratio/min": 0.25249814987182617, "sampling/sampling_logp_difference/max": 1.3763513565063477, "sampling/sampling_logp_difference/mean": 0.02781507931649685, "step": 2048 }, { "clip_ratio/high_max": 0.00854364933911711, "clip_ratio/high_mean": 0.00854364933911711, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/region_mean": 0.010255978093482554, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 72.875, "completions/mean_terminated_length": 72.875, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.19552704505622387, "epoch": 0.08229907217737077, "frac_reward_zero_std": 0.0, "grad_norm": 3.169921636581421, "learning_rate": 3.793939393939394e-06, "loss": 0.0002, "num_tokens": 4610199.0, "reward": 0.9960318207740784, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9960318207740784, "reward_meter_std": 0.0027530721854418516, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0027530903462320566, "reward_total_composite_mean": 0.9960318207740784, "reward_total_composite_std": 0.0027530721854418516, "reward_total_mean": 0.9960318207740784, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9960318207740784, "rewards/meter/std": 0.0027530721854418516, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9960318207740784, "rewards/total_composite/std": 0.0027530721854418516, "sampling/importance_sampling_ratio/max": 1.5734020471572876, "sampling/importance_sampling_ratio/mean": 1.0030927658081055, "sampling/importance_sampling_ratio/min": 0.3630704879760742, "sampling/sampling_logp_difference/max": 1.0131583213806152, "sampling/sampling_logp_difference/mean": 0.02808314375579357, "step": 2049 }, { "clip_ratio/high_max": 0.007843502098694444, "clip_ratio/high_mean": 0.007843502098694444, "clip_ratio/low_mean": 0.009891632944345474, "clip_ratio/low_min": 0.009891632944345474, "clip_ratio/region_mean": 0.017735135043039918, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.15631293877959251, "epoch": 0.08233923765915573, "frac_reward_zero_std": 0.0, "grad_norm": 2.9366605281829834, "learning_rate": 3.7909090909090914e-06, "loss": -0.0035, "num_tokens": 4612079.0, "reward": 0.8283913731575012, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8283913731575012, "reward_meter_std": 0.1856846958398819, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1856846958398819, "reward_total_composite_mean": 0.8283913731575012, "reward_total_composite_std": 0.1856846958398819, "reward_total_mean": 0.8283913731575012, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8283913731575012, "rewards/meter/std": 0.1856846958398819, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8283913731575012, "rewards/total_composite/std": 0.1856846958398819, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000585913658142, "sampling/importance_sampling_ratio/min": 0.09176904708147049, "sampling/sampling_logp_difference/max": 2.3884801864624023, "sampling/sampling_logp_difference/mean": 0.02338075079023838, "step": 2050 }, { "epoch": 0.08233923765915573, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 354.6923076923077, "eval_completions/max_terminated_length": 354.6923076923077, "eval_completions/mean_length": 201.2403846153846, "eval_completions/mean_terminated_length": 201.2403846153846, "eval_completions/min_length": 63.30769230769231, "eval_completions/min_terminated_length": 63.30769230769231, "eval_entropy": 0.12492357309047993, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 4612079.0, "eval_reward": 0.5263216747687414, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9102904154704168, "eval_reward_count_adherence_std": 0.12148115927210221, "eval_reward_meter_mean": 0.7046224291508014, "eval_reward_meter_std": 0.42778917917838466, "eval_reward_repeat_penalty_mean": 0.80578757249392, "eval_reward_repeat_penalty_std": 0.18289457318874505, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5263216747687414, "eval_reward_total_composite_std": 0.37407297583726734, "eval_reward_total_mean": 0.5263216747687414, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9102904154704168, "eval_rewards/count_adherence/std": 0.12148115927210221, "eval_rewards/meter/mean": 0.7046224291508014, "eval_rewards/meter/std": 0.42778917917838466, "eval_rewards/repeat_penalty/mean": 0.80578757249392, "eval_rewards/repeat_penalty/std": 0.18289457318874505, "eval_rewards/total_composite/mean": 0.5263216747687414, "eval_rewards/total_composite/std": 0.37407297583726734, "eval_runtime": 67.5758, "eval_samples_per_second": 1.539, "eval_sampling/importance_sampling_ratio/max": 1.4798904473964984, "eval_sampling/importance_sampling_ratio/mean": 1.0030618355824397, "eval_sampling/importance_sampling_ratio/min": 0.2842731796778165, "eval_sampling/sampling_logp_difference/max": 1.54466306246244, "eval_sampling/sampling_logp_difference/mean": 0.013937483756588055, "eval_steps_per_second": 0.192, "step": 2050 }, { "clip_ratio/high_max": 0.0008000020461622626, "clip_ratio/high_mean": 0.0008000020461622626, "clip_ratio/low_mean": 0.0003993610152974725, "clip_ratio/low_min": 0.0003993610152974725, "clip_ratio/region_mean": 0.001199363061459735, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 312.875, "completions/mean_terminated_length": 312.875, "completions/min_length": 312.0, "completions/min_terminated_length": 312.0, "entropy": 0.01078800146933645, "epoch": 0.08237940314094068, "frac_reward_zero_std": 0.0, "grad_norm": 0.02642267569899559, "learning_rate": 3.7878787878787882e-06, "loss": -0.0002, "num_tokens": 4616222.0, "reward": 0.5774303674697876, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973797798156738, "reward_meter_std": 8.498986062477343e-06, "reward_repeat_penalty_mean": 0.5789473652839661, "reward_repeat_penalty_std": 0.0, "reward_std": 4.94351661473047e-06, "reward_total_composite_mean": 0.5774303674697876, "reward_total_composite_std": 4.927758254780201e-06, "reward_total_mean": 0.5774303674697876, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973797798156738, "rewards/meter/std": 8.498986062477343e-06, "rewards/repeat_penalty/mean": 0.5789473652839661, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5774303674697876, "rewards/total_composite/std": 4.927758254780201e-06, "sampling/importance_sampling_ratio/max": 1.331079125404358, "sampling/importance_sampling_ratio/mean": 1.0007059574127197, "sampling/importance_sampling_ratio/min": 0.5910658836364746, "sampling/sampling_logp_difference/max": 0.5258277654647827, "sampling/sampling_logp_difference/mean": 0.001822733087465167, "step": 2051 }, { "clip_ratio/high_max": 0.014248464838601649, "clip_ratio/high_mean": 0.014248464838601649, "clip_ratio/low_mean": 0.0017639786819927394, "clip_ratio/low_min": 0.0017639786819927394, "clip_ratio/region_mean": 0.016012443520594388, "completions/clipped_ratio": 0.0, "completions/max_length": 221.0, "completions/max_terminated_length": 221.0, "completions/mean_length": 214.875, "completions/mean_terminated_length": 214.875, "completions/min_length": 208.0, "completions/min_terminated_length": 208.0, "entropy": 0.18392128869891167, "epoch": 0.08241956862272563, "frac_reward_zero_std": 0.0, "grad_norm": 1.5723775625228882, "learning_rate": 3.784848484848485e-06, "loss": -0.0126, "num_tokens": 4619661.0, "reward": 0.6123065948486328, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9847596883773804, "reward_meter_std": 0.03142433241009712, "reward_repeat_penalty_mean": 0.625, "reward_repeat_penalty_std": 0.2591308057308197, "reward_std": 0.24669425189495087, "reward_total_composite_mean": 0.6123065948486328, "reward_total_composite_std": 0.24669423699378967, "reward_total_mean": 0.6123065948486328, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9847596883773804, "rewards/meter/std": 0.03142433241009712, "rewards/repeat_penalty/mean": 0.625, "rewards/repeat_penalty/std": 0.2591308057308197, "rewards/total_composite/mean": 0.6123065948486328, "rewards/total_composite/std": 0.24669423699378967, "sampling/importance_sampling_ratio/max": 1.9927300214767456, "sampling/importance_sampling_ratio/mean": 1.0024995803833008, "sampling/importance_sampling_ratio/min": 0.14886726438999176, "sampling/sampling_logp_difference/max": 1.9047002792358398, "sampling/sampling_logp_difference/mean": 0.023984558880329132, "step": 2052 }, { "clip_ratio/high_max": 0.0065972222946584225, "clip_ratio/high_mean": 0.0065972222946584225, "clip_ratio/low_mean": 0.006253908621147275, "clip_ratio/low_min": 0.006253908621147275, "clip_ratio/region_mean": 0.012851130915805697, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 38.75, "completions/mean_terminated_length": 38.75, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.16460048034787178, "epoch": 0.08245973410451059, "frac_reward_zero_std": 0.0, "grad_norm": 7.329649448394775, "learning_rate": 3.781818181818182e-06, "loss": 0.0254, "num_tokens": 4621203.0, "reward": 0.9991681575775146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991681575775146, "reward_meter_std": 0.0005326797836460173, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005326886312104762, "reward_total_composite_mean": 0.9991681575775146, "reward_total_composite_std": 0.0005326797836460173, "reward_total_mean": 0.9991681575775146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991681575775146, "rewards/meter/std": 0.0005326797836460173, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991681575775146, "rewards/total_composite/std": 0.0005326797836460173, "sampling/importance_sampling_ratio/max": 1.5577260255813599, "sampling/importance_sampling_ratio/mean": 1.006339430809021, "sampling/importance_sampling_ratio/min": 0.4110788106918335, "sampling/sampling_logp_difference/max": 0.8889703750610352, "sampling/sampling_logp_difference/mean": 0.02681351639330387, "step": 2053 }, { "clip_ratio/high_max": 0.007575757801532745, "clip_ratio/high_mean": 0.007575757801532745, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/region_mean": 0.01515151560306549, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.03993460722267628, "epoch": 0.08249989958629554, "frac_reward_zero_std": 0.0, "grad_norm": 0.35728779435157776, "learning_rate": 3.778787878787879e-06, "loss": 0.0004, "num_tokens": 4623051.0, "reward": 0.997982382774353, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997982382774353, "reward_meter_std": 1.786020766303409e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.7854295947472565e-05, "reward_total_composite_mean": 0.997982382774353, "reward_total_composite_std": 1.786020766303409e-05, "reward_total_mean": 0.997982382774353, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997982382774353, "rewards/meter/std": 1.786020766303409e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997982382774353, "rewards/total_composite/std": 1.786020766303409e-05, "sampling/importance_sampling_ratio/max": 1.3692970275878906, "sampling/importance_sampling_ratio/mean": 0.999168872833252, "sampling/importance_sampling_ratio/min": 0.5883116722106934, "sampling/sampling_logp_difference/max": 0.5304985046386719, "sampling/sampling_logp_difference/mean": 0.009083742275834084, "step": 2054 }, { "clip_ratio/high_max": 0.026429779594764113, "clip_ratio/high_mean": 0.026429779594764113, "clip_ratio/low_mean": 0.006354167824611068, "clip_ratio/low_min": 0.006354167824611068, "clip_ratio/region_mean": 0.03278394741937518, "completions/clipped_ratio": 0.0, "completions/max_length": 330.0, "completions/max_terminated_length": 330.0, "completions/mean_length": 315.625, "completions/mean_terminated_length": 315.625, "completions/min_length": 305.0, "completions/min_terminated_length": 305.0, "entropy": 0.22647083643823862, "epoch": 0.0825400650680805, "frac_reward_zero_std": 0.0, "grad_norm": 2.26758074760437, "learning_rate": 3.775757575757576e-06, "loss": -0.0003, "num_tokens": 4627144.0, "reward": 0.6760411262512207, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.862500011920929, "reward_count_adherence_std": 0.0517548993229866, "reward_meter_mean": 0.9283133149147034, "reward_meter_std": 0.12819351255893707, "reward_repeat_penalty_mean": 0.8436580896377563, "reward_repeat_penalty_std": 0.11612330377101898, "reward_std": 0.1399042010307312, "reward_total_composite_mean": 0.6760411262512207, "reward_total_composite_std": 0.1399042010307312, "reward_total_mean": 0.6760411262512207, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.862500011920929, "rewards/count_adherence/std": 0.0517548993229866, "rewards/meter/mean": 0.9283133149147034, "rewards/meter/std": 0.12819351255893707, "rewards/repeat_penalty/mean": 0.8436580896377563, "rewards/repeat_penalty/std": 0.11612330377101898, "rewards/total_composite/mean": 0.6760411262512207, "rewards/total_composite/std": 0.1399042010307312, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002432942390442, "sampling/importance_sampling_ratio/min": 0.061752062290906906, "sampling/sampling_logp_difference/max": 2.784627914428711, "sampling/sampling_logp_difference/mean": 0.0354769341647625, "step": 2055 }, { "clip_ratio/high_max": 0.013611778849735856, "clip_ratio/high_mean": 0.013611778849735856, "clip_ratio/low_mean": 0.007753314450383186, "clip_ratio/low_min": 0.007753314450383186, "clip_ratio/region_mean": 0.021365093300119042, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 61.5, "completions/mean_terminated_length": 61.5, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.2539948094636202, "epoch": 0.08258023054986545, "frac_reward_zero_std": 0.0, "grad_norm": 8.520014762878418, "learning_rate": 3.772727272727273e-06, "loss": -0.0133, "num_tokens": 4628924.0, "reward": 0.6134400963783264, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.6756709814071655, "reward_meter_std": 0.25546279549598694, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22521549463272095, "reward_total_composite_mean": 0.6134400963783264, "reward_total_composite_std": 0.22521547973155975, "reward_total_mean": 0.6134400963783264, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.6756709814071655, "rewards/meter/std": 0.25546279549598694, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6134400963783264, "rewards/total_composite/std": 0.22521547973155975, "sampling/importance_sampling_ratio/max": 1.7200733423233032, "sampling/importance_sampling_ratio/mean": 1.0012702941894531, "sampling/importance_sampling_ratio/min": 0.10895545780658722, "sampling/sampling_logp_difference/max": 2.2168161869049072, "sampling/sampling_logp_difference/mean": 0.031044768169522285, "step": 2056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.00615574762923643, "epoch": 0.0826203960316504, "frac_reward_zero_std": 0.0, "grad_norm": 0.4782918691635132, "learning_rate": 3.76969696969697e-06, "loss": -0.001, "num_tokens": 4630892.0, "reward": 0.9990358352661133, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990358352661133, "reward_meter_std": 6.781429692637175e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.782029959140345e-05, "reward_total_composite_mean": 0.9990358352661133, "reward_total_composite_std": 6.781429692637175e-05, "reward_total_mean": 0.9990358352661133, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990358352661133, "rewards/meter/std": 6.781429692637175e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990358352661133, "rewards/total_composite/std": 6.781429692637175e-05, "sampling/importance_sampling_ratio/max": 1.0163952112197876, "sampling/importance_sampling_ratio/mean": 0.9997783303260803, "sampling/importance_sampling_ratio/min": 0.6607828736305237, "sampling/sampling_logp_difference/max": 0.41433000564575195, "sampling/sampling_logp_difference/mean": 0.0013893973082304, "step": 2057 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.029094983357936144, "epoch": 0.08266056151343536, "frac_reward_zero_std": 0.0, "grad_norm": 0.47224369645118713, "learning_rate": 3.766666666666667e-06, "loss": 0.0, "num_tokens": 4632844.0, "reward": 0.9979783296585083, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979783296585083, "reward_meter_std": 4.64778822788503e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.6468860091408715e-05, "reward_total_composite_mean": 0.9979783296585083, "reward_total_composite_std": 4.64778822788503e-05, "reward_total_mean": 0.9979783296585083, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979783296585083, "rewards/meter/std": 4.64778822788503e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979783296585083, "rewards/total_composite/std": 4.64778822788503e-05, "sampling/importance_sampling_ratio/max": 1.5034801959991455, "sampling/importance_sampling_ratio/mean": 1.0018833875656128, "sampling/importance_sampling_ratio/min": 0.6291179060935974, "sampling/sampling_logp_difference/max": 0.4634366035461426, "sampling/sampling_logp_difference/mean": 0.004593671765178442, "step": 2058 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.0075465834233909845, "epoch": 0.08270072699522031, "frac_reward_zero_std": 0.0, "grad_norm": 3.3012750148773193, "learning_rate": 3.7636363636363637e-06, "loss": -0.0039, "num_tokens": 4634771.0, "reward": 0.9989724159240723, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989724159240723, "reward_meter_std": 0.00024742307141423225, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002474260691087693, "reward_total_composite_mean": 0.9989724159240723, "reward_total_composite_std": 0.00024742307141423225, "reward_total_mean": 0.9989724159240723, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989724159240723, "rewards/meter/std": 0.00024742307141423225, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989724159240723, "rewards/total_composite/std": 0.00024742307141423225, "sampling/importance_sampling_ratio/max": 1.0218313932418823, "sampling/importance_sampling_ratio/mean": 0.9990096688270569, "sampling/importance_sampling_ratio/min": 0.2665814757347107, "sampling/sampling_logp_difference/max": 1.322075366973877, "sampling/sampling_logp_difference/mean": 0.0033799612428992987, "step": 2059 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/region_mean": 0.0025510203558951616, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 98.75, "completions/mean_terminated_length": 98.75, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.011058209463953972, "epoch": 0.08274089247700527, "frac_reward_zero_std": 0.0, "grad_norm": 0.7633002400398254, "learning_rate": 3.7606060606060605e-06, "loss": -0.0013, "num_tokens": 4636937.0, "reward": 0.99910569190979, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99910569190979, "reward_meter_std": 4.124943006900139e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.125862324144691e-05, "reward_total_composite_mean": 0.99910569190979, "reward_total_composite_std": 4.124943006900139e-05, "reward_total_mean": 0.99910569190979, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99910569190979, "rewards/meter/std": 4.124943006900139e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99910569190979, "rewards/total_composite/std": 4.124943006900139e-05, "sampling/importance_sampling_ratio/max": 1.0576543807983398, "sampling/importance_sampling_ratio/mean": 1.0008070468902588, "sampling/importance_sampling_ratio/min": 0.9609670042991638, "sampling/sampling_logp_difference/max": 0.05605363845825195, "sampling/sampling_logp_difference/mean": 0.0010279221460223198, "step": 2060 }, { "clip_ratio/high_max": 0.016136696096509695, "clip_ratio/high_mean": 0.016136696096509695, "clip_ratio/low_mean": 0.012820512987673283, "clip_ratio/low_min": 0.012820512987673283, "clip_ratio/region_mean": 0.028957209084182978, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 38.625, "completions/mean_terminated_length": 38.625, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.1621959926560521, "epoch": 0.08278105795879022, "frac_reward_zero_std": 0.0, "grad_norm": 2.503458023071289, "learning_rate": 3.757575757575758e-06, "loss": 0.0046, "num_tokens": 4638510.0, "reward": 0.9993641972541809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993641972541809, "reward_meter_std": 0.0002709543623495847, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002709643158596009, "reward_total_composite_mean": 0.9993641972541809, "reward_total_composite_std": 0.0002709543623495847, "reward_total_mean": 0.9993641972541809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993641972541809, "rewards/meter/std": 0.0002709543623495847, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993641972541809, "rewards/total_composite/std": 0.0002709543623495847, "sampling/importance_sampling_ratio/max": 1.691472053527832, "sampling/importance_sampling_ratio/mean": 1.0021777153015137, "sampling/importance_sampling_ratio/min": 0.3281702995300293, "sampling/sampling_logp_difference/max": 1.114222526550293, "sampling/sampling_logp_difference/mean": 0.026116767898201942, "step": 2061 }, { "clip_ratio/high_max": 0.0029450664296746254, "clip_ratio/high_mean": 0.0029450664296746254, "clip_ratio/low_mean": 0.0009765625, "clip_ratio/low_min": 0.0009765625, "clip_ratio/region_mean": 0.003921628929674625, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 127.875, "completions/mean_terminated_length": 127.875, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.019355892902240157, "epoch": 0.08282122344057517, "frac_reward_zero_std": 0.0, "grad_norm": 0.12022405117750168, "learning_rate": 3.7545454545454546e-06, "loss": 0.0007, "num_tokens": 4640965.0, "reward": 0.8553425073623657, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978995323181152, "reward_meter_std": 1.4925647519703489e-05, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 1.2798592251783703e-05, "reward_total_composite_mean": 0.8553425073623657, "reward_total_composite_std": 1.2786828847310971e-05, "reward_total_mean": 0.8553425073623657, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978995323181152, "rewards/meter/std": 1.4925647519703489e-05, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8553425073623657, "rewards/total_composite/std": 1.2786828847310971e-05, "sampling/importance_sampling_ratio/max": 1.108210802078247, "sampling/importance_sampling_ratio/mean": 0.9991137981414795, "sampling/importance_sampling_ratio/min": 0.44032585620880127, "sampling/sampling_logp_difference/max": 0.8202402591705322, "sampling/sampling_logp_difference/mean": 0.004213304258882999, "step": 2062 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 97.0, "completions/max_terminated_length": 97.0, "completions/mean_length": 97.0, "completions/mean_terminated_length": 97.0, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.020256070652976632, "epoch": 0.08286138892236013, "frac_reward_zero_std": 0.0, "grad_norm": 0.1908394694328308, "learning_rate": 3.7515151515151515e-06, "loss": -0.0003, "num_tokens": 4643037.0, "reward": 0.9979283809661865, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979283809661865, "reward_meter_std": 1.0387539987277705e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.038749087456381e-05, "reward_total_composite_mean": 0.9979283809661865, "reward_total_composite_std": 1.0387539987277705e-05, "reward_total_mean": 0.9979283809661865, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979283809661865, "rewards/meter/std": 1.0387539987277705e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979283809661865, "rewards/total_composite/std": 1.0387539987277705e-05, "sampling/importance_sampling_ratio/max": 1.0733686685562134, "sampling/importance_sampling_ratio/mean": 1.001339077949524, "sampling/importance_sampling_ratio/min": 0.5038148760795593, "sampling/sampling_logp_difference/max": 0.6855463981628418, "sampling/sampling_logp_difference/mean": 0.0028961896896362305, "step": 2063 }, { "clip_ratio/high_max": 0.0010416667209938169, "clip_ratio/high_mean": 0.0010416667209938169, "clip_ratio/low_mean": 0.004228386213071644, "clip_ratio/low_min": 0.004228386213071644, "clip_ratio/region_mean": 0.005270052934065461, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 119.625, "completions/mean_terminated_length": 119.625, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "entropy": 0.02747240220196545, "epoch": 0.08290155440414508, "frac_reward_zero_std": 0.0, "grad_norm": 0.6916006207466125, "learning_rate": 3.748484848484849e-06, "loss": -0.0028, "num_tokens": 4645538.0, "reward": 0.8543277978897095, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996715784072876, "reward_meter_std": 0.00013060594210401177, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011193443788215518, "reward_total_composite_mean": 0.8543277978897095, "reward_total_composite_std": 0.00011193905083928257, "reward_total_mean": 0.8543277978897095, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996715784072876, "rewards/meter/std": 0.00013060594210401177, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8543277978897095, "rewards/total_composite/std": 0.00011193905083928257, "sampling/importance_sampling_ratio/max": 1.5008399486541748, "sampling/importance_sampling_ratio/mean": 1.0017560720443726, "sampling/importance_sampling_ratio/min": 0.6449611186981201, "sampling/sampling_logp_difference/max": 0.4385652542114258, "sampling/sampling_logp_difference/mean": 0.004635249730199575, "step": 2064 }, { "clip_ratio/high_max": 0.016551562468521297, "clip_ratio/high_mean": 0.016551562468521297, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/region_mean": 0.02002378471661359, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 74.5, "completions/mean_terminated_length": 74.5, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.1982443816959858, "epoch": 0.08294171988593003, "frac_reward_zero_std": 0.0, "grad_norm": 3.2012574672698975, "learning_rate": 3.745454545454546e-06, "loss": -0.0076, "num_tokens": 4647558.0, "reward": 0.9918162822723389, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9918162822723389, "reward_meter_std": 0.009919598698616028, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009919599629938602, "reward_total_composite_mean": 0.9918162822723389, "reward_total_composite_std": 0.009919598698616028, "reward_total_mean": 0.9918162822723389, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9918162822723389, "rewards/meter/std": 0.009919598698616028, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9918162822723389, "rewards/total_composite/std": 0.009919598698616028, "sampling/importance_sampling_ratio/max": 1.9737117290496826, "sampling/importance_sampling_ratio/mean": 1.004847526550293, "sampling/importance_sampling_ratio/min": 0.33815622329711914, "sampling/sampling_logp_difference/max": 1.084247350692749, "sampling/sampling_logp_difference/mean": 0.023357173427939415, "step": 2065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.002455963665852323, "epoch": 0.08298188536771499, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.742424242424243e-06, "loss": 0.0, "num_tokens": 4649430.0, "reward": 0.9990598559379578, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9990598559379578, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0042266845703125, "sampling/importance_sampling_ratio/mean": 1.0002522468566895, "sampling/importance_sampling_ratio/min": 0.997229278087616, "sampling/sampling_logp_difference/max": 0.004217715933918953, "sampling/sampling_logp_difference/mean": 0.0002762091171462089, "step": 2066 }, { "clip_ratio/high_max": 0.012662693159654737, "clip_ratio/high_mean": 0.012662693159654737, "clip_ratio/low_mean": 0.009894335642457008, "clip_ratio/low_min": 0.009894335642457008, "clip_ratio/region_mean": 0.022557028802111745, "completions/clipped_ratio": 0.0, "completions/max_length": 398.0, "completions/max_terminated_length": 398.0, "completions/mean_length": 381.625, "completions/mean_terminated_length": 381.625, "completions/min_length": 362.0, "completions/min_terminated_length": 362.0, "entropy": 0.2074881698936224, "epoch": 0.08302205084949994, "frac_reward_zero_std": 0.0, "grad_norm": 1.5572625398635864, "learning_rate": 3.73939393939394e-06, "loss": -0.0169, "num_tokens": 4654027.0, "reward": 0.4973776340484619, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.65625, "reward_count_adherence_std": 0.033407654613256454, "reward_meter_mean": 0.9989120960235596, "reward_meter_std": 0.0005744565860368311, "reward_repeat_penalty_mean": 0.7585839033126831, "reward_repeat_penalty_std": 0.06556817144155502, "reward_std": 0.04983276501297951, "reward_total_composite_mean": 0.4973776340484619, "reward_total_composite_std": 0.04983275756239891, "reward_total_mean": 0.4973776340484619, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.65625, "rewards/count_adherence/std": 0.033407654613256454, "rewards/meter/mean": 0.9989120960235596, "rewards/meter/std": 0.0005744565860368311, "rewards/repeat_penalty/mean": 0.7585839033126831, "rewards/repeat_penalty/std": 0.06556817144155502, "rewards/total_composite/mean": 0.4973776340484619, "rewards/total_composite/std": 0.04983275756239891, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0025993585586548, "sampling/importance_sampling_ratio/min": 4.6404053932747047e-07, "sampling/sampling_logp_difference/max": 14.583293914794922, "sampling/sampling_logp_difference/mean": 0.034968629479408264, "step": 2067 }, { "clip_ratio/high_max": 0.01815969729796052, "clip_ratio/high_mean": 0.01815969729796052, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.021731125889346004, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 74.875, "completions/mean_terminated_length": 74.875, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.22592910565435886, "epoch": 0.0830622163312849, "frac_reward_zero_std": 0.0, "grad_norm": 5.365983009338379, "learning_rate": 3.736363636363637e-06, "loss": -0.0084, "num_tokens": 4655794.0, "reward": 0.9989867210388184, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989867210388184, "reward_meter_std": 0.0009042008896358311, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000904214452020824, "reward_total_composite_mean": 0.9989867210388184, "reward_total_composite_std": 0.0009042008896358311, "reward_total_mean": 0.9989867210388184, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989867210388184, "rewards/meter/std": 0.0009042008896358311, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989867210388184, "rewards/total_composite/std": 0.0009042008896358311, "sampling/importance_sampling_ratio/max": 1.8772411346435547, "sampling/importance_sampling_ratio/mean": 1.0074291229248047, "sampling/importance_sampling_ratio/min": 0.44946879148483276, "sampling/sampling_logp_difference/max": 0.7996888160705566, "sampling/sampling_logp_difference/mean": 0.027786457911133766, "step": 2068 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 29.75, "completions/mean_terminated_length": 29.75, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.03380009834654629, "epoch": 0.08310238181306985, "frac_reward_zero_std": 0.0, "grad_norm": 3.824120283126831, "learning_rate": 3.7333333333333337e-06, "loss": 0.008, "num_tokens": 4657320.0, "reward": 0.9928488731384277, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9928488731384277, "reward_meter_std": 7.0027461333666e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.002745405770838e-05, "reward_total_composite_mean": 0.9928488731384277, "reward_total_composite_std": 7.0027461333666e-05, "reward_total_mean": 0.9928488731384277, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9928488731384277, "rewards/meter/std": 7.0027461333666e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9928488731384277, "rewards/total_composite/std": 7.0027461333666e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008081316947937, "sampling/importance_sampling_ratio/min": 0.9426794052124023, "sampling/sampling_logp_difference/max": 0.6952643394470215, "sampling/sampling_logp_difference/mean": 0.0071687763556838036, "step": 2069 }, { "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/low_mean": 0.008620689623057842, "clip_ratio/low_min": 0.008620689623057842, "clip_ratio/region_mean": 0.010852832579985261, "completions/clipped_ratio": 0.0, "completions/max_length": 58.0, "completions/max_terminated_length": 58.0, "completions/mean_length": 57.5, "completions/mean_terminated_length": 57.5, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.03393339877948165, "epoch": 0.0831425472948548, "frac_reward_zero_std": 0.0, "grad_norm": 2.178863286972046, "learning_rate": 3.7303030303030306e-06, "loss": 0.0041, "num_tokens": 4659028.0, "reward": 0.9949300289154053, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949300289154053, "reward_meter_std": 4.729199281428009e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.729198190034367e-05, "reward_total_composite_mean": 0.9949300289154053, "reward_total_composite_std": 4.729199281428009e-05, "reward_total_mean": 0.9949300289154053, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949300289154053, "rewards/meter/std": 4.729199281428009e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949300289154053, "rewards/total_composite/std": 4.729199281428009e-05, "sampling/importance_sampling_ratio/max": 1.951530933380127, "sampling/importance_sampling_ratio/mean": 1.0030012130737305, "sampling/importance_sampling_ratio/min": 0.5056865215301514, "sampling/sampling_logp_difference/max": 0.6818382740020752, "sampling/sampling_logp_difference/mean": 0.006639817729592323, "step": 2070 }, { "clip_ratio/high_max": 0.011511339340358973, "clip_ratio/high_mean": 0.011511339340358973, "clip_ratio/low_mean": 0.0028572361916303635, "clip_ratio/low_min": 0.0028572361916303635, "clip_ratio/region_mean": 0.014368575531989336, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 87.25, "completions/mean_terminated_length": 87.25, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.06696707848459482, "epoch": 0.08318271277663976, "frac_reward_zero_std": 0.0, "grad_norm": 2.837266206741333, "learning_rate": 3.727272727272728e-06, "loss": 0.0012, "num_tokens": 4661038.0, "reward": 0.996016263961792, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996016263961792, "reward_meter_std": 0.00032200998975895345, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0003220097569283098, "reward_total_composite_mean": 0.996016263961792, "reward_total_composite_std": 0.00032200998975895345, "reward_total_mean": 0.996016263961792, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996016263961792, "rewards/meter/std": 0.00032200998975895345, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996016263961792, "rewards/total_composite/std": 0.00032200998975895345, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0012880563735962, "sampling/importance_sampling_ratio/min": 0.3910037875175476, "sampling/sampling_logp_difference/max": 1.0829687118530273, "sampling/sampling_logp_difference/mean": 0.015046549029648304, "step": 2071 }, { "clip_ratio/high_max": 0.003246753243729472, "clip_ratio/high_mean": 0.003246753243729472, "clip_ratio/low_mean": 0.014854058739729226, "clip_ratio/low_min": 0.014854058739729226, "clip_ratio/region_mean": 0.018100811983458698, "completions/clipped_ratio": 0.0, "completions/max_length": 156.0, "completions/max_terminated_length": 156.0, "completions/mean_length": 151.875, "completions/mean_terminated_length": 151.875, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.11779375281184912, "epoch": 0.08322287825842471, "frac_reward_zero_std": 0.0, "grad_norm": 3.793785572052002, "learning_rate": 3.7242424242424246e-06, "loss": 0.003, "num_tokens": 4663829.0, "reward": 0.8616158962249756, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9847034215927124, "reward_meter_std": 0.01465265266597271, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05130419507622719, "reward_total_composite_mean": 0.8616158962249756, "reward_total_composite_std": 0.05130418762564659, "reward_total_mean": 0.8616158962249756, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9847034215927124, "rewards/meter/std": 0.01465265266597271, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8616158962249756, "rewards/total_composite/std": 0.05130418762564659, "sampling/importance_sampling_ratio/max": 1.8727295398712158, "sampling/importance_sampling_ratio/mean": 1.002585530281067, "sampling/importance_sampling_ratio/min": 0.11767642945051193, "sampling/sampling_logp_difference/max": 2.1398165225982666, "sampling/sampling_logp_difference/mean": 0.02519957348704338, "step": 2072 }, { "clip_ratio/high_max": 0.023355348501354456, "clip_ratio/high_mean": 0.023355348501354456, "clip_ratio/low_mean": 0.009154040599241853, "clip_ratio/low_min": 0.009154040599241853, "clip_ratio/region_mean": 0.03250938910059631, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 112.25, "completions/mean_terminated_length": 112.25, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.3353299479931593, "epoch": 0.08326304374020967, "frac_reward_zero_std": 0.0, "grad_norm": 3.991828203201294, "learning_rate": 3.7212121212121215e-06, "loss": -0.0078, "num_tokens": 4665887.0, "reward": 0.99208664894104, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99208664894104, "reward_meter_std": 0.012512140907347202, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.012512128800153732, "reward_total_composite_mean": 0.99208664894104, "reward_total_composite_std": 0.012512140907347202, "reward_total_mean": 0.99208664894104, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99208664894104, "rewards/meter/std": 0.012512140907347202, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99208664894104, "rewards/total_composite/std": 0.012512140907347202, "sampling/importance_sampling_ratio/max": 1.9161767959594727, "sampling/importance_sampling_ratio/mean": 1.0028138160705566, "sampling/importance_sampling_ratio/min": 0.23588424921035767, "sampling/sampling_logp_difference/max": 1.4444141387939453, "sampling/sampling_logp_difference/mean": 0.04193767532706261, "step": 2073 }, { "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/region_mean": 0.003968254197388887, "completions/clipped_ratio": 0.0, "completions/max_length": 63.0, "completions/max_terminated_length": 63.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.02836236241273582, "epoch": 0.08330320922199462, "frac_reward_zero_std": 0.0, "grad_norm": 1.4791640043258667, "learning_rate": 3.7181818181818187e-06, "loss": 0.0016, "num_tokens": 4667687.0, "reward": 0.9969918131828308, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969918131828308, "reward_meter_std": 0.00010383435437688604, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010383195331087336, "reward_total_composite_mean": 0.9969918131828308, "reward_total_composite_std": 0.00010383435437688604, "reward_total_mean": 0.9969918131828308, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969918131828308, "rewards/meter/std": 0.00010383435437688604, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969918131828308, "rewards/total_composite/std": 0.00010383435437688604, "sampling/importance_sampling_ratio/max": 1.7541072368621826, "sampling/importance_sampling_ratio/mean": 1.0029563903808594, "sampling/importance_sampling_ratio/min": 0.7295204997062683, "sampling/sampling_logp_difference/max": 0.561959981918335, "sampling/sampling_logp_difference/mean": 0.005412998143583536, "step": 2074 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.0028630376618821174, "epoch": 0.08334337470377957, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.7151515151515156e-06, "loss": 0.0, "num_tokens": 4669527.0, "reward": 0.9990598559379578, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9990598559379578, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0042898654937744, "sampling/importance_sampling_ratio/mean": 1.0002044439315796, "sampling/importance_sampling_ratio/min": 0.9917435646057129, "sampling/sampling_logp_difference/max": 0.008290700614452362, "sampling/sampling_logp_difference/mean": 0.0002672372793313116, "step": 2075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.005342192598618567, "clip_ratio/low_min": 0.005342192598618567, "clip_ratio/region_mean": 0.005342192598618567, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 116.75, "completions/mean_terminated_length": 116.75, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.04457000829279423, "epoch": 0.08338354018556453, "frac_reward_zero_std": 0.0, "grad_norm": 2.183760404586792, "learning_rate": 3.7121212121212124e-06, "loss": -0.0008, "num_tokens": 4671885.0, "reward": 0.8721410036087036, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996731698513031, "reward_meter_std": 3.716555511346087e-05, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05035989359021187, "reward_total_composite_mean": 0.8721410036087036, "reward_total_composite_std": 0.050359878689050674, "reward_total_mean": 0.8721410036087036, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996731698513031, "rewards/meter/std": 3.716555511346087e-05, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8721410036087036, "rewards/total_composite/std": 0.050359878689050674, "sampling/importance_sampling_ratio/max": 1.463352918624878, "sampling/importance_sampling_ratio/mean": 1.000087022781372, "sampling/importance_sampling_ratio/min": 0.5082767009735107, "sampling/sampling_logp_difference/max": 0.6767293214797974, "sampling/sampling_logp_difference/mean": 0.008479885756969452, "step": 2076 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/region_mean": 0.0037313431967049837, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.75, "completions/mean_terminated_length": 68.75, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.05782870203256607, "epoch": 0.08342370566734948, "frac_reward_zero_std": 0.0, "grad_norm": 5.309196949005127, "learning_rate": 3.7090909090909092e-06, "loss": -0.005, "num_tokens": 4673787.0, "reward": 0.9131183624267578, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9131183624267578, "reward_meter_std": 0.07230029255151749, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.07230029255151749, "reward_total_composite_mean": 0.9131183624267578, "reward_total_composite_std": 0.07230029255151749, "reward_total_mean": 0.9131183624267578, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9131183624267578, "rewards/meter/std": 0.07230029255151749, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9131183624267578, "rewards/total_composite/std": 0.07230029255151749, "sampling/importance_sampling_ratio/max": 1.6235750913619995, "sampling/importance_sampling_ratio/mean": 1.0044362545013428, "sampling/importance_sampling_ratio/min": 0.6702300310134888, "sampling/sampling_logp_difference/max": 0.4846305847167969, "sampling/sampling_logp_difference/mean": 0.007340815383940935, "step": 2077 }, { "clip_ratio/high_max": 0.00663501548115164, "clip_ratio/high_mean": 0.00663501548115164, "clip_ratio/low_mean": 0.006791699095629156, "clip_ratio/low_min": 0.006791699095629156, "clip_ratio/region_mean": 0.013426714576780796, "completions/clipped_ratio": 0.0, "completions/max_length": 153.0, "completions/max_terminated_length": 153.0, "completions/mean_length": 148.375, "completions/mean_terminated_length": 148.375, "completions/min_length": 146.0, "completions/min_terminated_length": 146.0, "entropy": 0.20128737576305866, "epoch": 0.08346387114913444, "frac_reward_zero_std": 0.0, "grad_norm": 2.607077121734619, "learning_rate": 3.7060606060606065e-06, "loss": -0.0024, "num_tokens": 4676302.0, "reward": 0.909817099571228, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990102648735046, "reward_meter_std": 0.00021416397066786885, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.0739244818687439, "reward_total_composite_mean": 0.909817099571228, "reward_total_composite_std": 0.0739244893193245, "reward_total_mean": 0.909817099571228, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990102648735046, "rewards/meter/std": 0.00021416397066786885, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.909817099571228, "rewards/total_composite/std": 0.0739244893193245, "sampling/importance_sampling_ratio/max": 1.6944489479064941, "sampling/importance_sampling_ratio/mean": 1.0051114559173584, "sampling/importance_sampling_ratio/min": 0.26833122968673706, "sampling/sampling_logp_difference/max": 1.31553316116333, "sampling/sampling_logp_difference/mean": 0.023411214351654053, "step": 2078 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.00562528264708817, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.026706934673711658, "epoch": 0.08350403663091939, "frac_reward_zero_std": 0.0, "grad_norm": 0.3224610984325409, "learning_rate": 3.7030303030303033e-06, "loss": -0.0002, "num_tokens": 4678159.0, "reward": 0.9980389475822449, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980389475822449, "reward_meter_std": 2.6020999939646572e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.602726817713119e-05, "reward_total_composite_mean": 0.9980389475822449, "reward_total_composite_std": 2.6020999939646572e-05, "reward_total_mean": 0.9980389475822449, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980389475822449, "rewards/meter/std": 2.6020999939646572e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980389475822449, "rewards/total_composite/std": 2.6020999939646572e-05, "sampling/importance_sampling_ratio/max": 1.2167119979858398, "sampling/importance_sampling_ratio/mean": 0.9994921088218689, "sampling/importance_sampling_ratio/min": 0.3345133364200592, "sampling/sampling_logp_difference/max": 1.095078468322754, "sampling/sampling_logp_difference/mean": 0.007125611882656813, "step": 2079 }, { "clip_ratio/high_max": 0.03190367738716304, "clip_ratio/high_mean": 0.03190367738716304, "clip_ratio/low_mean": 0.02227564202621579, "clip_ratio/low_min": 0.02227564202621579, "clip_ratio/region_mean": 0.054179319413378835, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 39.625, "completions/mean_terminated_length": 39.625, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.395177373662591, "epoch": 0.08354420211270434, "frac_reward_zero_std": 0.0, "grad_norm": 9.717095375061035, "learning_rate": 3.7e-06, "loss": -0.0081, "num_tokens": 4679892.0, "reward": 0.9970946311950684, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970946311950684, "reward_meter_std": 0.003818577155470848, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0038185729645192623, "reward_total_composite_mean": 0.9970946311950684, "reward_total_composite_std": 0.003818577155470848, "reward_total_mean": 0.9970946311950684, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970946311950684, "rewards/meter/std": 0.003818577155470848, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970946311950684, "rewards/total_composite/std": 0.003818577155470848, "sampling/importance_sampling_ratio/max": 1.6322083473205566, "sampling/importance_sampling_ratio/mean": 0.9983224868774414, "sampling/importance_sampling_ratio/min": 0.2037201076745987, "sampling/sampling_logp_difference/max": 1.591008186340332, "sampling/sampling_logp_difference/mean": 0.058814503252506256, "step": 2080 }, { "clip_ratio/high_max": 0.009914012742228806, "clip_ratio/high_mean": 0.009914012742228806, "clip_ratio/low_mean": 0.00844594556838274, "clip_ratio/low_min": 0.00844594556838274, "clip_ratio/region_mean": 0.018359958310611546, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 75.625, "completions/mean_terminated_length": 75.625, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.22988875024020672, "epoch": 0.0835843675944893, "frac_reward_zero_std": 0.0, "grad_norm": 7.125683784484863, "learning_rate": 3.6969696969696974e-06, "loss": -0.0053, "num_tokens": 4681889.0, "reward": 0.8755629062652588, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8755629062652588, "reward_meter_std": 0.3493680953979492, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3493681252002716, "reward_total_composite_mean": 0.8755629062652588, "reward_total_composite_std": 0.3493680953979492, "reward_total_mean": 0.8755629062652588, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8755629062652588, "rewards/meter/std": 0.3493680953979492, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8755629062652588, "rewards/total_composite/std": 0.3493680953979492, "sampling/importance_sampling_ratio/max": 1.5550665855407715, "sampling/importance_sampling_ratio/mean": 1.0054106712341309, "sampling/importance_sampling_ratio/min": 0.3693508207798004, "sampling/sampling_logp_difference/max": 0.9960083961486816, "sampling/sampling_logp_difference/mean": 0.032194141298532486, "step": 2081 }, { "clip_ratio/high_max": 0.001923076924867928, "clip_ratio/high_mean": 0.001923076924867928, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.001923076924867928, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.026136466301977634, "epoch": 0.08362453307627425, "frac_reward_zero_std": 0.0, "grad_norm": 0.7917746305465698, "learning_rate": 3.6939393939393942e-06, "loss": -0.0013, "num_tokens": 4683848.0, "reward": 0.9979181289672852, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979181289672852, "reward_meter_std": 0.00028089084662497044, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00028088997351005673, "reward_total_composite_mean": 0.9979181289672852, "reward_total_composite_std": 0.00028089084662497044, "reward_total_mean": 0.9979181289672852, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979181289672852, "rewards/meter/std": 0.00028089084662497044, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979181289672852, "rewards/total_composite/std": 0.00028089084662497044, "sampling/importance_sampling_ratio/max": 1.159103274345398, "sampling/importance_sampling_ratio/mean": 0.9990794658660889, "sampling/importance_sampling_ratio/min": 0.3388626277446747, "sampling/sampling_logp_difference/max": 1.082160472869873, "sampling/sampling_logp_difference/mean": 0.008248316124081612, "step": 2082 }, { "clip_ratio/high_max": 0.03633964783512056, "clip_ratio/high_mean": 0.03633964783512056, "clip_ratio/low_mean": 0.0051369862630963326, "clip_ratio/low_min": 0.0051369862630963326, "clip_ratio/region_mean": 0.04147663409821689, "completions/clipped_ratio": 0.0, "completions/max_length": 150.0, "completions/max_terminated_length": 150.0, "completions/mean_length": 147.5, "completions/mean_terminated_length": 147.5, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.2962766755372286, "epoch": 0.0836646985580592, "frac_reward_zero_std": 0.0, "grad_norm": 3.155012845993042, "learning_rate": 3.690909090909091e-06, "loss": 0.0021, "num_tokens": 4686452.0, "reward": 0.9801425337791443, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979666471481323, "reward_meter_std": 0.0005141000729054213, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.050341930240392685, "reward_total_composite_mean": 0.9801425337791443, "reward_total_composite_std": 0.05034194886684418, "reward_total_mean": 0.9801425337791443, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979666471481323, "rewards/meter/std": 0.0005141000729054213, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9801425337791443, "rewards/total_composite/std": 0.05034194886684418, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0034173727035522, "sampling/importance_sampling_ratio/min": 0.27929648756980896, "sampling/sampling_logp_difference/max": 1.2754813432693481, "sampling/sampling_logp_difference/mean": 0.039336614310741425, "step": 2083 }, { "clip_ratio/high_max": 0.020977655542083085, "clip_ratio/high_mean": 0.020977655542083085, "clip_ratio/low_mean": 0.004023639601655304, "clip_ratio/low_min": 0.004023639601655304, "clip_ratio/region_mean": 0.02500129514373839, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 151.875, "completions/mean_terminated_length": 151.875, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.14115788787603378, "epoch": 0.08370486403984416, "frac_reward_zero_std": 0.0, "grad_norm": 5.227519989013672, "learning_rate": 3.687878787878788e-06, "loss": 0.0021, "num_tokens": 4689035.0, "reward": 0.7994784116744995, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8921954035758972, "reward_meter_std": 0.1858879029750824, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.18709427118301392, "reward_total_composite_mean": 0.7994784116744995, "reward_total_composite_std": 0.18709427118301392, "reward_total_mean": 0.7994784116744995, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8921954035758972, "rewards/meter/std": 0.1858879029750824, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7994784116744995, "rewards/total_composite/std": 0.18709427118301392, "sampling/importance_sampling_ratio/max": 1.7797260284423828, "sampling/importance_sampling_ratio/mean": 1.0021029710769653, "sampling/importance_sampling_ratio/min": 0.06448182463645935, "sampling/sampling_logp_difference/max": 2.7413718700408936, "sampling/sampling_logp_difference/mean": 0.031672801822423935, "step": 2084 }, { "clip_ratio/high_max": 0.003448275849223137, "clip_ratio/high_mean": 0.003448275849223137, "clip_ratio/low_mean": 0.018270427826792, "clip_ratio/low_min": 0.018270427826792, "clip_ratio/region_mean": 0.02171870367601514, "completions/clipped_ratio": 0.0, "completions/max_length": 310.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 294.125, "completions/mean_terminated_length": 294.125, "completions/min_length": 288.0, "completions/min_terminated_length": 288.0, "entropy": 0.2046606820076704, "epoch": 0.08374502952162911, "frac_reward_zero_std": 0.0, "grad_norm": 1.964224934577942, "learning_rate": 3.684848484848485e-06, "loss": 0.0076, "num_tokens": 4693004.0, "reward": 0.8750678896903992, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998891294002533, "reward_meter_std": 0.0004529697762336582, "reward_repeat_penalty_mean": 0.8760416507720947, "reward_repeat_penalty_std": 0.023332269862294197, "reward_std": 0.023195745423436165, "reward_total_composite_mean": 0.8750678896903992, "reward_total_composite_std": 0.02319573611021042, "reward_total_mean": 0.8750678896903992, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998891294002533, "rewards/meter/std": 0.0004529697762336582, "rewards/repeat_penalty/mean": 0.8760416507720947, "rewards/repeat_penalty/std": 0.023332269862294197, "rewards/total_composite/mean": 0.8750678896903992, "rewards/total_composite/std": 0.02319573611021042, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0067555904388428, "sampling/importance_sampling_ratio/min": 0.1141689196228981, "sampling/sampling_logp_difference/max": 2.1700761318206787, "sampling/sampling_logp_difference/mean": 0.025770394131541252, "step": 2085 }, { "clip_ratio/high_max": 0.020660461392253637, "clip_ratio/high_mean": 0.020660461392253637, "clip_ratio/low_mean": 0.007962123490869999, "clip_ratio/low_min": 0.007962123490869999, "clip_ratio/region_mean": 0.028622584883123636, "completions/clipped_ratio": 0.0, "completions/max_length": 306.0, "completions/max_terminated_length": 306.0, "completions/mean_length": 291.125, "completions/mean_terminated_length": 291.125, "completions/min_length": 270.0, "completions/min_terminated_length": 270.0, "entropy": 0.23322945460677147, "epoch": 0.08378519500341407, "frac_reward_zero_std": 0.0, "grad_norm": 1.5824189186096191, "learning_rate": 3.681818181818182e-06, "loss": -0.0218, "num_tokens": 4696909.0, "reward": 0.8602873086929321, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9986486434936523, "reward_meter_std": 0.0003730578755494207, "reward_repeat_penalty_mean": 0.8737351298332214, "reward_repeat_penalty_std": 0.06906846165657043, "reward_std": 0.09172171354293823, "reward_total_composite_mean": 0.8602873086929321, "reward_total_composite_std": 0.09172172844409943, "reward_total_mean": 0.8602873086929321, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9986486434936523, "rewards/meter/std": 0.0003730578755494207, "rewards/repeat_penalty/mean": 0.8737351298332214, "rewards/repeat_penalty/std": 0.06906846165657043, "rewards/total_composite/mean": 0.8602873086929321, "rewards/total_composite/std": 0.09172172844409943, "sampling/importance_sampling_ratio/max": 1.9022753238677979, "sampling/importance_sampling_ratio/mean": 1.003360390663147, "sampling/importance_sampling_ratio/min": 1.240273945768422e-07, "sampling/sampling_logp_difference/max": 15.902763366699219, "sampling/sampling_logp_difference/mean": 0.0357806496322155, "step": 2086 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.0027300505607854575, "epoch": 0.08382536048519902, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.678787878787879e-06, "loss": 0.0, "num_tokens": 4698765.0, "reward": 0.9990598559379578, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9990598559379578, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0067503452301025, "sampling/importance_sampling_ratio/mean": 1.0002596378326416, "sampling/importance_sampling_ratio/min": 0.9977940917015076, "sampling/sampling_logp_difference/max": 0.006727563217282295, "sampling/sampling_logp_difference/mean": 0.000278584222542122, "step": 2087 }, { "clip_ratio/high_max": 0.013125053141266108, "clip_ratio/high_mean": 0.013125053141266108, "clip_ratio/low_mean": 0.007895297603681684, "clip_ratio/low_min": 0.007895297603681684, "clip_ratio/region_mean": 0.02102035074494779, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 112.125, "completions/mean_terminated_length": 112.125, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.18240004498511553, "epoch": 0.08386552596698398, "frac_reward_zero_std": 0.0, "grad_norm": 3.470649480819702, "learning_rate": 3.6757575757575757e-06, "loss": -0.0178, "num_tokens": 4701006.0, "reward": 0.9967917203903198, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9967917203903198, "reward_meter_std": 0.002291632117703557, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002291641663759947, "reward_total_composite_mean": 0.9967917203903198, "reward_total_composite_std": 0.002291632117703557, "reward_total_mean": 0.9967917203903198, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9967917203903198, "rewards/meter/std": 0.002291632117703557, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967917203903198, "rewards/total_composite/std": 0.002291632117703557, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0074383020401, "sampling/importance_sampling_ratio/min": 0.3058212697505951, "sampling/sampling_logp_difference/max": 1.1847543716430664, "sampling/sampling_logp_difference/mean": 0.02775193564593792, "step": 2088 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.0029864944226574153, "epoch": 0.08390569144876893, "frac_reward_zero_std": 0.0, "grad_norm": 1.3091340065002441, "learning_rate": 3.672727272727273e-06, "loss": -0.0005, "num_tokens": 4702766.0, "reward": 0.9990187883377075, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990187883377075, "reward_meter_std": 0.00011603027814999223, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011603629536693916, "reward_total_composite_mean": 0.9990187883377075, "reward_total_composite_std": 0.00011603027814999223, "reward_total_mean": 0.9990187883377075, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990187883377075, "rewards/meter/std": 0.00011603027814999223, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990187883377075, "rewards/total_composite/std": 0.00011603027814999223, "sampling/importance_sampling_ratio/max": 1.0052270889282227, "sampling/importance_sampling_ratio/mean": 0.9990017414093018, "sampling/importance_sampling_ratio/min": 0.3340601921081543, "sampling/sampling_logp_difference/max": 1.0964341163635254, "sampling/sampling_logp_difference/mean": 0.0023684005718678236, "step": 2089 }, { "clip_ratio/high_max": 0.0100741779897362, "clip_ratio/high_mean": 0.0100741779897362, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/region_mean": 0.011636678013019264, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 86.0, "completions/mean_terminated_length": 86.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.05126679269596934, "epoch": 0.08394585693055388, "frac_reward_zero_std": 0.0, "grad_norm": 2.0700643062591553, "learning_rate": 3.6696969696969697e-06, "loss": -0.0218, "num_tokens": 4704838.0, "reward": 0.9953513741493225, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953513741493225, "reward_meter_std": 0.0020426749251782894, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0020426807459443808, "reward_total_composite_mean": 0.9953513741493225, "reward_total_composite_std": 0.0020426749251782894, "reward_total_mean": 0.9953513741493225, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953513741493225, "rewards/meter/std": 0.0020426749251782894, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953513741493225, "rewards/total_composite/std": 0.0020426749251782894, "sampling/importance_sampling_ratio/max": 1.8694008588790894, "sampling/importance_sampling_ratio/mean": 1.0013517141342163, "sampling/importance_sampling_ratio/min": 0.288541316986084, "sampling/sampling_logp_difference/max": 1.2429170608520508, "sampling/sampling_logp_difference/mean": 0.013093557208776474, "step": 2090 }, { "clip_ratio/high_max": 0.017056229757145047, "clip_ratio/high_mean": 0.017056229757145047, "clip_ratio/low_mean": 0.0033445945591665804, "clip_ratio/low_min": 0.0033445945591665804, "clip_ratio/region_mean": 0.020400824316311628, "completions/clipped_ratio": 0.0, "completions/max_length": 150.0, "completions/max_terminated_length": 150.0, "completions/mean_length": 147.0, "completions/mean_terminated_length": 147.0, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.19478315208107233, "epoch": 0.08398602241233884, "frac_reward_zero_std": 0.0, "grad_norm": 2.3087821006774902, "learning_rate": 3.6666666666666666e-06, "loss": 0.0105, "num_tokens": 4707486.0, "reward": 0.959259033203125, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949081540107727, "reward_meter_std": 0.004385494627058506, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.0641113668680191, "reward_total_composite_mean": 0.959259033203125, "reward_total_composite_std": 0.0641113817691803, "reward_total_mean": 0.959259033203125, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949081540107727, "rewards/meter/std": 0.004385494627058506, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.959259033203125, "rewards/total_composite/std": 0.0641113817691803, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0080565214157104, "sampling/importance_sampling_ratio/min": 0.3291736841201782, "sampling/sampling_logp_difference/max": 1.1111698150634766, "sampling/sampling_logp_difference/mean": 0.027278585359454155, "step": 2091 }, { "clip_ratio/high_max": 0.012833506800234318, "clip_ratio/high_mean": 0.012833506800234318, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/region_mean": 0.016211885260418057, "completions/clipped_ratio": 0.0, "completions/max_length": 40.0, "completions/max_terminated_length": 40.0, "completions/mean_length": 38.75, "completions/mean_terminated_length": 38.75, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.1498966608196497, "epoch": 0.08402618789412379, "frac_reward_zero_std": 0.0, "grad_norm": 6.707172393798828, "learning_rate": 3.6636363636363643e-06, "loss": 0.0061, "num_tokens": 4709052.0, "reward": 0.9994734525680542, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994734525680542, "reward_meter_std": 0.0001241376594407484, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00012412367505021393, "reward_total_composite_mean": 0.9994734525680542, "reward_total_composite_std": 0.0001241376594407484, "reward_total_mean": 0.9994734525680542, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994734525680542, "rewards/meter/std": 0.0001241376594407484, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994734525680542, "rewards/total_composite/std": 0.0001241376594407484, "sampling/importance_sampling_ratio/max": 1.5661958456039429, "sampling/importance_sampling_ratio/mean": 0.9975427389144897, "sampling/importance_sampling_ratio/min": 0.3822647035121918, "sampling/sampling_logp_difference/max": 0.9616420269012451, "sampling/sampling_logp_difference/mean": 0.031195173040032387, "step": 2092 }, { "clip_ratio/high_max": 0.006756756920367479, "clip_ratio/high_mean": 0.006756756920367479, "clip_ratio/low_mean": 0.009615384740754962, "clip_ratio/low_min": 0.009615384740754962, "clip_ratio/region_mean": 0.01637214166112244, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 37.75, "completions/mean_terminated_length": 37.75, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.12450700253248215, "epoch": 0.08406635337590875, "frac_reward_zero_std": 0.0, "grad_norm": 7.211493015289307, "learning_rate": 3.660606060606061e-06, "loss": 0.0237, "num_tokens": 4710458.0, "reward": 0.9790675640106201, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9790675640106201, "reward_meter_std": 0.02823478914797306, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02823478914797306, "reward_total_composite_mean": 0.9790675640106201, "reward_total_composite_std": 0.02823478914797306, "reward_total_mean": 0.9790675640106201, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9790675640106201, "rewards/meter/std": 0.02823478914797306, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9790675640106201, "rewards/total_composite/std": 0.02823478914797306, "sampling/importance_sampling_ratio/max": 1.3320231437683105, "sampling/importance_sampling_ratio/mean": 1.0054656267166138, "sampling/importance_sampling_ratio/min": 0.7474565505981445, "sampling/sampling_logp_difference/max": 0.29107916355133057, "sampling/sampling_logp_difference/mean": 0.014203748665750027, "step": 2093 }, { "clip_ratio/high_max": 0.0036231884732842445, "clip_ratio/high_mean": 0.0036231884732842445, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0036231884732842445, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.04171428643167019, "epoch": 0.0841065188576937, "frac_reward_zero_std": 0.0, "grad_norm": 0.3633219003677368, "learning_rate": 3.657575757575758e-06, "loss": 0.0017, "num_tokens": 4712186.0, "reward": 0.938814103603363, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.938814103603363, "reward_meter_std": 0.0002475241490174085, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002475333458278328, "reward_total_composite_mean": 0.938814103603363, "reward_total_composite_std": 0.0002475241490174085, "reward_total_mean": 0.938814103603363, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.938814103603363, "rewards/meter/std": 0.0002475241490174085, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.938814103603363, "rewards/total_composite/std": 0.0002475241490174085, "sampling/importance_sampling_ratio/max": 1.0647722482681274, "sampling/importance_sampling_ratio/mean": 0.999586820602417, "sampling/importance_sampling_ratio/min": 0.4818035662174225, "sampling/sampling_logp_difference/max": 0.7302188873291016, "sampling/sampling_logp_difference/mean": 0.0060532353818416595, "step": 2094 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.00596828549169004, "epoch": 0.08414668433947865, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.654545454545455e-06, "loss": 0.0, "num_tokens": 4714042.0, "reward": 0.9990598559379578, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990598559379578, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9990598559379578, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9990598559379578, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990598559379578, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990598559379578, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0163171291351318, "sampling/importance_sampling_ratio/mean": 1.0003678798675537, "sampling/importance_sampling_ratio/min": 0.9876941442489624, "sampling/sampling_logp_difference/max": 0.016185477375984192, "sampling/sampling_logp_difference/mean": 0.0005561705911532044, "step": 2095 }, { "clip_ratio/high_max": 0.0036231884732842445, "clip_ratio/high_mean": 0.0036231884732842445, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/region_mean": 0.005434782709926367, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.07018918683752418, "epoch": 0.0841868498212636, "frac_reward_zero_std": 0.0, "grad_norm": 3.099059581756592, "learning_rate": 3.651515151515152e-06, "loss": 0.006, "num_tokens": 4716098.0, "reward": 0.9360407590866089, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9360407590866089, "reward_meter_std": 0.005241453181952238, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0052414704114198685, "reward_total_composite_mean": 0.9360407590866089, "reward_total_composite_std": 0.005241453181952238, "reward_total_mean": 0.9360407590866089, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9360407590866089, "rewards/meter/std": 0.005241453181952238, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9360407590866089, "rewards/total_composite/std": 0.005241453181952238, "sampling/importance_sampling_ratio/max": 1.8257942199707031, "sampling/importance_sampling_ratio/mean": 1.001249074935913, "sampling/importance_sampling_ratio/min": 0.49197322130203247, "sampling/sampling_logp_difference/max": 0.7093310356140137, "sampling/sampling_logp_difference/mean": 0.011832116171717644, "step": 2096 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.030742917326278985, "epoch": 0.08422701530304856, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.648484848484849e-06, "loss": 0.0, "num_tokens": 4717538.0, "reward": 0.9942078590393066, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942078590393066, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9942078590393066, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9942078590393066, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942078590393066, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942078590393066, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0437800884246826, "sampling/importance_sampling_ratio/mean": 1.0009886026382446, "sampling/importance_sampling_ratio/min": 0.8580098152160645, "sampling/sampling_logp_difference/max": 0.15313977003097534, "sampling/sampling_logp_difference/mean": 0.0031366986222565174, "step": 2097 }, { "clip_ratio/high_max": 0.028136852895841002, "clip_ratio/high_mean": 0.028136852895841002, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/region_mean": 0.029699352919124067, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 75.625, "completions/mean_terminated_length": 75.625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.16028152592480183, "epoch": 0.08426718078483351, "frac_reward_zero_std": 0.0, "grad_norm": 2.8510286808013916, "learning_rate": 3.645454545454546e-06, "loss": 0.0272, "num_tokens": 4719239.0, "reward": 0.9827600717544556, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9827600717544556, "reward_meter_std": 0.0294841006398201, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.029484104365110397, "reward_total_composite_mean": 0.9827600717544556, "reward_total_composite_std": 0.0294841006398201, "reward_total_mean": 0.9827600717544556, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9827600717544556, "rewards/meter/std": 0.0294841006398201, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9827600717544556, "rewards/total_composite/std": 0.0294841006398201, "sampling/importance_sampling_ratio/max": 1.6294512748718262, "sampling/importance_sampling_ratio/mean": 0.9956676363945007, "sampling/importance_sampling_ratio/min": 0.3161216974258423, "sampling/sampling_logp_difference/max": 1.151628017425537, "sampling/sampling_logp_difference/mean": 0.033739130944013596, "step": 2098 }, { "clip_ratio/high_max": 0.016345139825716615, "clip_ratio/high_mean": 0.016345139825716615, "clip_ratio/low_mean": 0.008760618977248669, "clip_ratio/low_min": 0.008760618977248669, "clip_ratio/region_mean": 0.025105758802965283, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 105.5, "completions/mean_terminated_length": 105.5, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.29543587751686573, "epoch": 0.08430734626661847, "frac_reward_zero_std": 0.0, "grad_norm": 4.330075740814209, "learning_rate": 3.642424242424243e-06, "loss": -0.0241, "num_tokens": 4721395.0, "reward": 0.8491353988647461, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8654639720916748, "reward_meter_std": 0.23051653802394867, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.25138649344444275, "reward_total_composite_mean": 0.8491353988647461, "reward_total_composite_std": 0.25138652324676514, "reward_total_mean": 0.8491353988647461, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8654639720916748, "rewards/meter/std": 0.23051653802394867, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8491353988647461, "rewards/total_composite/std": 0.25138652324676514, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007623553276062, "sampling/importance_sampling_ratio/min": 0.3179101049900055, "sampling/sampling_logp_difference/max": 1.3270950317382812, "sampling/sampling_logp_difference/mean": 0.03153692185878754, "step": 2099 }, { "clip_ratio/high_max": 0.008620689623057842, "clip_ratio/high_mean": 0.008620689623057842, "clip_ratio/low_mean": 0.0060240961611270905, "clip_ratio/low_min": 0.0060240961611270905, "clip_ratio/region_mean": 0.014644785784184933, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 86.5, "completions/mean_terminated_length": 86.5, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.20394995296373963, "epoch": 0.08434751174840342, "frac_reward_zero_std": 0.0, "grad_norm": 7.931367874145508, "learning_rate": 3.6393939393939398e-06, "loss": 0.0052, "num_tokens": 4723535.0, "reward": 0.9882360100746155, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9882360100746155, "reward_meter_std": 0.022600755095481873, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.022600755095481873, "reward_total_composite_mean": 0.9882360100746155, "reward_total_composite_std": 0.022600755095481873, "reward_total_mean": 0.9882360100746155, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9882360100746155, "rewards/meter/std": 0.022600755095481873, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9882360100746155, "rewards/total_composite/std": 0.022600755095481873, "sampling/importance_sampling_ratio/max": 1.8274431228637695, "sampling/importance_sampling_ratio/mean": 1.0013149976730347, "sampling/importance_sampling_ratio/min": 0.22358666360378265, "sampling/sampling_logp_difference/max": 1.4979562759399414, "sampling/sampling_logp_difference/mean": 0.022525055333971977, "step": 2100 }, { "epoch": 0.08434751174840342, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 356.53846153846155, "eval_completions/max_terminated_length": 356.53846153846155, "eval_completions/mean_length": 201.65384615384616, "eval_completions/mean_terminated_length": 201.65384615384616, "eval_completions/min_length": 61.92307692307692, "eval_completions/min_terminated_length": 61.92307692307692, "eval_entropy": 0.12711333655394041, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 4723535.0, "eval_reward": 0.542255353469115, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9138026375036973, "eval_reward_count_adherence_std": 0.12064289301633835, "eval_reward_meter_mean": 0.7131731968659621, "eval_reward_meter_std": 0.4128540616769057, "eval_reward_repeat_penalty_mean": 0.8164694079985986, "eval_reward_repeat_penalty_std": 0.1486260168827497, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.542255353469115, "eval_reward_total_composite_std": 0.3623279837461618, "eval_reward_total_mean": 0.542255353469115, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9138026375036973, "eval_rewards/count_adherence/std": 0.12064289301633835, "eval_rewards/meter/mean": 0.7131731968659621, "eval_rewards/meter/std": 0.4128540616769057, "eval_rewards/repeat_penalty/mean": 0.8164694079985986, "eval_rewards/repeat_penalty/std": 0.1486260168827497, "eval_rewards/total_composite/mean": 0.542255353469115, "eval_rewards/total_composite/std": 0.3623279837461618, "eval_runtime": 68.5013, "eval_samples_per_second": 1.518, "eval_sampling/importance_sampling_ratio/max": 1.501889733167795, "eval_sampling/importance_sampling_ratio/mean": 1.0036665751383855, "eval_sampling/importance_sampling_ratio/min": 0.38750078357183015, "eval_sampling/sampling_logp_difference/max": 0.9667405440257146, "eval_sampling/sampling_logp_difference/mean": 0.013091834047092842, "eval_steps_per_second": 0.19, "step": 2100 }, { "clip_ratio/high_max": 0.00595723238075152, "clip_ratio/high_mean": 0.00595723238075152, "clip_ratio/low_mean": 0.007700952875893563, "clip_ratio/low_min": 0.007700952875893563, "clip_ratio/region_mean": 0.013658185256645083, "completions/clipped_ratio": 0.0, "completions/max_length": 192.0, "completions/max_terminated_length": 192.0, "completions/mean_length": 176.375, "completions/mean_terminated_length": 176.375, "completions/min_length": 160.0, "completions/min_terminated_length": 160.0, "entropy": 0.11573350336402655, "epoch": 0.08438767723018838, "frac_reward_zero_std": 0.0, "grad_norm": 1.2342872619628906, "learning_rate": 3.6363636363636366e-06, "loss": -0.0478, "num_tokens": 4726322.0, "reward": 0.6860010623931885, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.1035098284482956, "reward_meter_mean": 0.9766891598701477, "reward_meter_std": 0.022879159078001976, "reward_repeat_penalty_mean": 0.761904776096344, "reward_repeat_penalty_std": 0.07493293285369873, "reward_std": 0.08363301306962967, "reward_total_composite_mean": 0.6860010623931885, "reward_total_composite_std": 0.08363299816846848, "reward_total_mean": 0.6860010623931885, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.1035098284482956, "rewards/meter/mean": 0.9766891598701477, "rewards/meter/std": 0.022879159078001976, "rewards/repeat_penalty/mean": 0.761904776096344, "rewards/repeat_penalty/std": 0.07493293285369873, "rewards/total_composite/mean": 0.6860010623931885, "rewards/total_composite/std": 0.08363299816846848, "sampling/importance_sampling_ratio/max": 1.9307383298873901, "sampling/importance_sampling_ratio/mean": 1.0021946430206299, "sampling/importance_sampling_ratio/min": 0.36357343196868896, "sampling/sampling_logp_difference/max": 1.0117740631103516, "sampling/sampling_logp_difference/mean": 0.01796703413128853, "step": 2101 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 28.875, "completions/mean_terminated_length": 28.875, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.03053366276435554, "epoch": 0.08442784271197333, "frac_reward_zero_std": 0.0, "grad_norm": 14.514649391174316, "learning_rate": 3.633333333333334e-06, "loss": 0.0086, "num_tokens": 4727945.0, "reward": 0.9928563833236694, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9928563833236694, "reward_meter_std": 0.0002995376707985997, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002995316463056952, "reward_total_composite_mean": 0.9928563833236694, "reward_total_composite_std": 0.0002995376707985997, "reward_total_mean": 0.9928563833236694, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9928563833236694, "rewards/meter/std": 0.0002995376707985997, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9928563833236694, "rewards/total_composite/std": 0.0002995376707985997, "sampling/importance_sampling_ratio/max": 1.1580135822296143, "sampling/importance_sampling_ratio/mean": 1.0009772777557373, "sampling/importance_sampling_ratio/min": 0.8847241401672363, "sampling/sampling_logp_difference/max": 0.14670616388320923, "sampling/sampling_logp_difference/mean": 0.0030352873727679253, "step": 2102 }, { "clip_ratio/high_max": 0.02126871724613011, "clip_ratio/high_mean": 0.02126871724613011, "clip_ratio/low_mean": 0.00048638132284395397, "clip_ratio/low_min": 0.00048638132284395397, "clip_ratio/region_mean": 0.021755098568974063, "completions/clipped_ratio": 0.0, "completions/max_length": 286.0, "completions/max_terminated_length": 286.0, "completions/mean_length": 279.25, "completions/mean_terminated_length": 279.25, "completions/min_length": 257.0, "completions/min_terminated_length": 257.0, "entropy": 0.272721977904439, "epoch": 0.08446800819375828, "frac_reward_zero_std": 0.0, "grad_norm": 1.2050065994262695, "learning_rate": 3.6303030303030307e-06, "loss": -0.0264, "num_tokens": 4731675.0, "reward": 0.8288908004760742, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9298574924468994, "reward_meter_std": 0.1897430717945099, "reward_repeat_penalty_mean": 0.8583333492279053, "reward_repeat_penalty_std": 0.19002924859523773, "reward_std": 0.2640323340892792, "reward_total_composite_mean": 0.8288908004760742, "reward_total_composite_std": 0.26403236389160156, "reward_total_mean": 0.8288908004760742, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9298574924468994, "rewards/meter/std": 0.1897430717945099, "rewards/repeat_penalty/mean": 0.8583333492279053, "rewards/repeat_penalty/std": 0.19002924859523773, "rewards/total_composite/mean": 0.8288908004760742, "rewards/total_composite/std": 0.26403236389160156, "sampling/importance_sampling_ratio/max": 1.7007434368133545, "sampling/importance_sampling_ratio/mean": 1.0039503574371338, "sampling/importance_sampling_ratio/min": 0.14267219603061676, "sampling/sampling_logp_difference/max": 1.9472055435180664, "sampling/sampling_logp_difference/mean": 0.023323198780417442, "step": 2103 }, { "clip_ratio/high_max": 0.009039964294061065, "clip_ratio/high_mean": 0.009039964294061065, "clip_ratio/low_mean": 0.004951660754159093, "clip_ratio/low_min": 0.004951660754159093, "clip_ratio/region_mean": 0.013991625048220158, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 296.5, "completions/mean_terminated_length": 296.5, "completions/min_length": 284.0, "completions/min_terminated_length": 284.0, "entropy": 0.17900856165215373, "epoch": 0.08450817367554324, "frac_reward_zero_std": 0.0, "grad_norm": 1.8657711744308472, "learning_rate": 3.6272727272727275e-06, "loss": -0.0122, "num_tokens": 4735591.0, "reward": 0.6847283840179443, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9669773578643799, "reward_meter_std": 0.07153221219778061, "reward_repeat_penalty_mean": 0.7205882668495178, "reward_repeat_penalty_std": 0.061355408281087875, "reward_std": 0.07457955926656723, "reward_total_composite_mean": 0.6847283840179443, "reward_total_composite_std": 0.07457956671714783, "reward_total_mean": 0.6847283840179443, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9669773578643799, "rewards/meter/std": 0.07153221219778061, "rewards/repeat_penalty/mean": 0.7205882668495178, "rewards/repeat_penalty/std": 0.061355408281087875, "rewards/total_composite/mean": 0.6847283840179443, "rewards/total_composite/std": 0.07457956671714783, "sampling/importance_sampling_ratio/max": 1.9632225036621094, "sampling/importance_sampling_ratio/mean": 1.0043342113494873, "sampling/importance_sampling_ratio/min": 0.3145197927951813, "sampling/sampling_logp_difference/max": 1.1567082405090332, "sampling/sampling_logp_difference/mean": 0.021201852709054947, "step": 2104 }, { "clip_ratio/high_max": 0.009068090235814452, "clip_ratio/high_mean": 0.009068090235814452, "clip_ratio/low_mean": 0.004629980307072401, "clip_ratio/low_min": 0.004629980307072401, "clip_ratio/region_mean": 0.013698070542886853, "completions/clipped_ratio": 0.0, "completions/max_length": 166.0, "completions/max_terminated_length": 166.0, "completions/mean_length": 164.75, "completions/mean_terminated_length": 164.75, "completions/min_length": 161.0, "completions/min_terminated_length": 161.0, "entropy": 0.0402666125446558, "epoch": 0.08454833915732819, "frac_reward_zero_std": 0.0, "grad_norm": 4.257723331451416, "learning_rate": 3.6242424242424248e-06, "loss": -0.0002, "num_tokens": 4738461.0, "reward": 0.7769712209701538, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989630579948425, "reward_meter_std": 0.00037122031790204346, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002887141308747232, "reward_total_composite_mean": 0.7769712209701538, "reward_total_composite_std": 0.00028871477115899324, "reward_total_mean": 0.7769712209701538, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989630579948425, "rewards/meter/std": 0.00037122031790204346, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7769712209701538, "rewards/total_composite/std": 0.00028871477115899324, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0012632608413696, "sampling/importance_sampling_ratio/min": 0.09893912822008133, "sampling/sampling_logp_difference/max": 2.3132505416870117, "sampling/sampling_logp_difference/mean": 0.016738269478082657, "step": 2105 }, { "clip_ratio/high_max": 0.00443893379997462, "clip_ratio/high_mean": 0.00443893379997462, "clip_ratio/low_mean": 0.00402287530596368, "clip_ratio/low_min": 0.00402287530596368, "clip_ratio/region_mean": 0.0084618091059383, "completions/clipped_ratio": 0.0, "completions/max_length": 282.0, "completions/max_terminated_length": 282.0, "completions/mean_length": 279.5, "completions/mean_terminated_length": 279.5, "completions/min_length": 276.0, "completions/min_terminated_length": 276.0, "entropy": 0.045266281347721815, "epoch": 0.08458850463911315, "frac_reward_zero_std": 0.0, "grad_norm": 1.7383067607879639, "learning_rate": 3.6212121212121216e-06, "loss": -0.0003, "num_tokens": 4742393.0, "reward": 0.7536386847496033, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9945741891860962, "reward_meter_std": 0.0010695032542571425, "reward_repeat_penalty_mean": 0.7578125, "reward_repeat_penalty_std": 0.07790146768093109, "reward_std": 0.07672013342380524, "reward_total_composite_mean": 0.7536386847496033, "reward_total_composite_std": 0.07672014087438583, "reward_total_mean": 0.7536386847496033, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9945741891860962, "rewards/meter/std": 0.0010695032542571425, "rewards/repeat_penalty/mean": 0.7578125, "rewards/repeat_penalty/std": 0.07790146768093109, "rewards/total_composite/mean": 0.7536386847496033, "rewards/total_composite/std": 0.07672014087438583, "sampling/importance_sampling_ratio/max": 1.8650121688842773, "sampling/importance_sampling_ratio/mean": 0.999245285987854, "sampling/importance_sampling_ratio/min": 0.2669290006160736, "sampling/sampling_logp_difference/max": 1.3207725286483765, "sampling/sampling_logp_difference/mean": 0.010709014721214771, "step": 2106 }, { "clip_ratio/high_max": 0.0010593220358714461, "clip_ratio/high_mean": 0.0010593220358714461, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/region_mean": 0.0031779661076143384, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 118.0, "completions/mean_terminated_length": 118.0, "completions/min_length": 118.0, "completions/min_terminated_length": 118.0, "entropy": 0.026277633383870125, "epoch": 0.0846286701208981, "frac_reward_zero_std": 0.0, "grad_norm": 0.0878043994307518, "learning_rate": 3.6181818181818184e-06, "loss": -0.0, "num_tokens": 4744793.0, "reward": 0.8544071912765503, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9968082904815674, "reward_meter_std": 4.3226536945439875e-05, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 3.7032186810392886e-05, "reward_total_composite_mean": 0.8544071912765503, "reward_total_composite_std": 3.7037829315522686e-05, "reward_total_mean": 0.8544071912765503, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9968082904815674, "rewards/meter/std": 4.3226536945439875e-05, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8544071912765503, "rewards/total_composite/std": 3.7037829315522686e-05, "sampling/importance_sampling_ratio/max": 1.1646240949630737, "sampling/importance_sampling_ratio/mean": 0.9996362924575806, "sampling/importance_sampling_ratio/min": 0.33887895941734314, "sampling/sampling_logp_difference/max": 1.0821123123168945, "sampling/sampling_logp_difference/mean": 0.005184296518564224, "step": 2107 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.004166666883975267, "clip_ratio/low_min": 0.004166666883975267, "clip_ratio/region_mean": 0.004166666883975267, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 120.5, "completions/mean_terminated_length": 120.5, "completions/min_length": 120.0, "completions/min_terminated_length": 120.0, "entropy": 0.029683739179745317, "epoch": 0.08466883560268305, "frac_reward_zero_std": 0.0, "grad_norm": 1.9401229619979858, "learning_rate": 3.6151515151515153e-06, "loss": -0.0069, "num_tokens": 4747165.0, "reward": 0.8397266268730164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9796810746192932, "reward_meter_std": 0.003423919202759862, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00293477950617671, "reward_total_composite_mean": 0.8397266268730164, "reward_total_composite_std": 0.002934773452579975, "reward_total_mean": 0.8397266268730164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9796810746192932, "rewards/meter/std": 0.003423919202759862, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8397266268730164, "rewards/total_composite/std": 0.002934773452579975, "sampling/importance_sampling_ratio/max": 1.215042233467102, "sampling/importance_sampling_ratio/mean": 1.0006637573242188, "sampling/importance_sampling_ratio/min": 0.761186420917511, "sampling/sampling_logp_difference/max": 0.2728769779205322, "sampling/sampling_logp_difference/mean": 0.004073810297995806, "step": 2108 }, { "clip_ratio/high_max": 0.006250000325962901, "clip_ratio/high_mean": 0.006250000325962901, "clip_ratio/low_mean": 0.021334989927709103, "clip_ratio/low_min": 0.021334989927709103, "clip_ratio/region_mean": 0.027584990253672004, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 59.25, "completions/mean_terminated_length": 59.25, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.24129995983093977, "epoch": 0.08470900108446801, "frac_reward_zero_std": 0.0, "grad_norm": 14.469276428222656, "learning_rate": 3.6121212121212125e-06, "loss": 0.0148, "num_tokens": 4748903.0, "reward": 0.675230860710144, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.675230860710144, "reward_meter_std": 0.3926205635070801, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3926205635070801, "reward_total_composite_mean": 0.675230860710144, "reward_total_composite_std": 0.3926205635070801, "reward_total_mean": 0.675230860710144, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.675230860710144, "rewards/meter/std": 0.3926205635070801, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.675230860710144, "rewards/total_composite/std": 0.3926205635070801, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008945107460022, "sampling/importance_sampling_ratio/min": 0.35194310545921326, "sampling/sampling_logp_difference/max": 1.3949341773986816, "sampling/sampling_logp_difference/mean": 0.032343991100788116, "step": 2109 }, { "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/region_mean": 0.004504870157688856, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 55.875, "completions/mean_terminated_length": 55.875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.031467124819755554, "epoch": 0.08474916656625296, "frac_reward_zero_std": 0.0, "grad_norm": 0.8487328290939331, "learning_rate": 3.6090909090909093e-06, "loss": -0.0044, "num_tokens": 4750550.0, "reward": 0.9948891401290894, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948891401290894, "reward_meter_std": 0.0002349100832361728, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00023490135208703578, "reward_total_composite_mean": 0.9948891401290894, "reward_total_composite_std": 0.0002349100832361728, "reward_total_mean": 0.9948891401290894, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948891401290894, "rewards/meter/std": 0.0002349100832361728, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948891401290894, "rewards/total_composite/std": 0.0002349100832361728, "sampling/importance_sampling_ratio/max": 1.5498037338256836, "sampling/importance_sampling_ratio/mean": 1.0012332201004028, "sampling/importance_sampling_ratio/min": 0.7602133750915527, "sampling/sampling_logp_difference/max": 0.4381282329559326, "sampling/sampling_logp_difference/mean": 0.005522139370441437, "step": 2110 }, { "clip_ratio/high_max": 0.022355403285473585, "clip_ratio/high_mean": 0.022355403285473585, "clip_ratio/low_mean": 0.012152777868323028, "clip_ratio/low_min": 0.012152777868323028, "clip_ratio/region_mean": 0.03450818115379661, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.125, "completions/mean_terminated_length": 72.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.2009260579943657, "epoch": 0.08478933204803792, "frac_reward_zero_std": 0.0, "grad_norm": 4.360041618347168, "learning_rate": 3.606060606060606e-06, "loss": 0.0071, "num_tokens": 4752319.0, "reward": 0.9952445030212402, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952445030212402, "reward_meter_std": 0.0019698443356901407, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00196986086666584, "reward_total_composite_mean": 0.9952445030212402, "reward_total_composite_std": 0.0019698443356901407, "reward_total_mean": 0.9952445030212402, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952445030212402, "rewards/meter/std": 0.0019698443356901407, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9952445030212402, "rewards/total_composite/std": 0.0019698443356901407, "sampling/importance_sampling_ratio/max": 1.8677667379379272, "sampling/importance_sampling_ratio/mean": 1.0026034116744995, "sampling/importance_sampling_ratio/min": 0.24193060398101807, "sampling/sampling_logp_difference/max": 1.4191043376922607, "sampling/sampling_logp_difference/mean": 0.039312709122896194, "step": 2111 }, { "clip_ratio/high_max": 0.013099490897729993, "clip_ratio/high_mean": 0.013099490897729993, "clip_ratio/low_mean": 0.009980987291783094, "clip_ratio/low_min": 0.009980987291783094, "clip_ratio/region_mean": 0.023080478189513087, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 75.5, "completions/mean_terminated_length": 75.5, "completions/min_length": 74.0, "completions/min_terminated_length": 74.0, "entropy": 0.28017595037817955, "epoch": 0.08482949752982287, "frac_reward_zero_std": 0.0, "grad_norm": 3.6561806201934814, "learning_rate": 3.603030303030303e-06, "loss": 0.0141, "num_tokens": 4754067.0, "reward": 0.9953248500823975, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953248500823975, "reward_meter_std": 0.002693862421438098, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002693843562155962, "reward_total_composite_mean": 0.9953248500823975, "reward_total_composite_std": 0.002693862421438098, "reward_total_mean": 0.9953248500823975, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953248500823975, "rewards/meter/std": 0.002693862421438098, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953248500823975, "rewards/total_composite/std": 0.002693862421438098, "sampling/importance_sampling_ratio/max": 1.8056482076644897, "sampling/importance_sampling_ratio/mean": 1.0075361728668213, "sampling/importance_sampling_ratio/min": 0.2211194485425949, "sampling/sampling_logp_difference/max": 1.5090522766113281, "sampling/sampling_logp_difference/mean": 0.0394962802529335, "step": 2112 }, { "clip_ratio/high_max": 0.0022863353369757533, "clip_ratio/high_mean": 0.0022863353369757533, "clip_ratio/low_mean": 0.0008680555620230734, "clip_ratio/low_min": 0.0008680555620230734, "clip_ratio/region_mean": 0.0031543908989988267, "completions/clipped_ratio": 0.0, "completions/max_length": 289.0, "completions/max_terminated_length": 289.0, "completions/mean_length": 283.875, "completions/mean_terminated_length": 283.875, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.03367242659442127, "epoch": 0.08486966301160782, "frac_reward_zero_std": 0.0, "grad_norm": 1.0902267694473267, "learning_rate": 3.6000000000000003e-06, "loss": 0.0042, "num_tokens": 4758194.0, "reward": 0.6962326169013977, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955472946166992, "reward_meter_std": 0.0001822304038796574, "reward_repeat_penalty_mean": 0.6993464231491089, "reward_repeat_penalty_std": 0.05026431009173393, "reward_std": 0.05004315450787544, "reward_total_composite_mean": 0.6962326169013977, "reward_total_composite_std": 0.05004315823316574, "reward_total_mean": 0.6962326169013977, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955472946166992, "rewards/meter/std": 0.0001822304038796574, "rewards/repeat_penalty/mean": 0.6993464231491089, "rewards/repeat_penalty/std": 0.05026431009173393, "rewards/total_composite/mean": 0.6962326169013977, "rewards/total_composite/std": 0.05004315823316574, "sampling/importance_sampling_ratio/max": 1.2696807384490967, "sampling/importance_sampling_ratio/mean": 0.9997674822807312, "sampling/importance_sampling_ratio/min": 0.4324061870574951, "sampling/sampling_logp_difference/max": 0.8383898735046387, "sampling/sampling_logp_difference/mean": 0.0065383305773139, "step": 2113 }, { "clip_ratio/high_max": 0.01766940689412877, "clip_ratio/high_mean": 0.01766940689412877, "clip_ratio/low_mean": 0.003332580265123397, "clip_ratio/low_min": 0.003332580265123397, "clip_ratio/region_mean": 0.021001987159252167, "completions/clipped_ratio": 0.0, "completions/max_length": 193.0, "completions/max_terminated_length": 193.0, "completions/mean_length": 184.375, "completions/mean_terminated_length": 184.375, "completions/min_length": 176.0, "completions/min_terminated_length": 176.0, "entropy": 0.16935125924646854, "epoch": 0.08490982849339278, "frac_reward_zero_std": 0.0, "grad_norm": 2.4666974544525146, "learning_rate": 3.596969696969697e-06, "loss": 0.005, "num_tokens": 4761069.0, "reward": 0.8585879802703857, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969959855079651, "reward_meter_std": 0.0019897068850696087, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.07856741547584534, "reward_std": 0.0791376456618309, "reward_total_composite_mean": 0.8585879802703857, "reward_total_composite_std": 0.0791376605629921, "reward_total_mean": 0.8585879802703857, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969959855079651, "rewards/meter/std": 0.0019897068850696087, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8585879802703857, "rewards/total_composite/std": 0.0791376605629921, "sampling/importance_sampling_ratio/max": 1.9149926900863647, "sampling/importance_sampling_ratio/mean": 1.0035723447799683, "sampling/importance_sampling_ratio/min": 0.3695699870586395, "sampling/sampling_logp_difference/max": 0.995415210723877, "sampling/sampling_logp_difference/mean": 0.0264764241874218, "step": 2114 }, { "clip_ratio/high_max": 0.013179366709664464, "clip_ratio/high_mean": 0.013179366709664464, "clip_ratio/low_mean": 0.008480485761538148, "clip_ratio/low_min": 0.008480485761538148, "clip_ratio/region_mean": 0.021659852471202612, "completions/clipped_ratio": 0.0, "completions/max_length": 399.0, "completions/max_terminated_length": 399.0, "completions/mean_length": 371.75, "completions/mean_terminated_length": 371.75, "completions/min_length": 346.0, "completions/min_terminated_length": 346.0, "entropy": 0.23509259428828955, "epoch": 0.08494999397517773, "frac_reward_zero_std": 0.0, "grad_norm": 1.3121997117996216, "learning_rate": 3.593939393939394e-06, "loss": -0.0201, "num_tokens": 4765867.0, "reward": 0.5668459534645081, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7142857313156128, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9931106567382812, "reward_meter_std": 0.003387609263882041, "reward_repeat_penalty_mean": 0.7990131378173828, "reward_repeat_penalty_std": 0.08971592038869858, "reward_std": 0.06429950892925262, "reward_total_composite_mean": 0.5668459534645081, "reward_total_composite_std": 0.06429950892925262, "reward_total_mean": 0.5668459534645081, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7142857313156128, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9931106567382812, "rewards/meter/std": 0.003387609263882041, "rewards/repeat_penalty/mean": 0.7990131378173828, "rewards/repeat_penalty/std": 0.08971592038869858, "rewards/total_composite/mean": 0.5668459534645081, "rewards/total_composite/std": 0.06429950892925262, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0030711889266968, "sampling/importance_sampling_ratio/min": 0.055317867547273636, "sampling/sampling_logp_difference/max": 2.8946592807769775, "sampling/sampling_logp_difference/mean": 0.03195931762456894, "step": 2115 }, { "clip_ratio/high_max": 0.0015060240402817726, "clip_ratio/high_mean": 0.0015060240402817726, "clip_ratio/low_mean": 0.0015151514671742916, "clip_ratio/low_min": 0.0015151514671742916, "clip_ratio/region_mean": 0.0030211755074560642, "completions/clipped_ratio": 0.0, "completions/max_length": 166.0, "completions/max_terminated_length": 166.0, "completions/mean_length": 165.125, "completions/mean_terminated_length": 165.125, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.04427940887399018, "epoch": 0.08499015945696269, "frac_reward_zero_std": 0.0, "grad_norm": 1.9204028844833374, "learning_rate": 3.590909090909091e-06, "loss": 0.0004, "num_tokens": 4768740.0, "reward": 0.8187738060951233, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991816878318787, "reward_meter_std": 9.469691576668993e-06, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.05745694413781166, "reward_total_composite_mean": 0.8187738060951233, "reward_total_composite_std": 0.057456936687231064, "reward_total_mean": 0.8187738060951233, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991816878318787, "rewards/meter/std": 9.469691576668993e-06, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.8187738060951233, "rewards/total_composite/std": 0.057456936687231064, "sampling/importance_sampling_ratio/max": 1.796520471572876, "sampling/importance_sampling_ratio/mean": 0.9997456669807434, "sampling/importance_sampling_ratio/min": 0.33717140555381775, "sampling/sampling_logp_difference/max": 1.0871639251708984, "sampling/sampling_logp_difference/mean": 0.009189879521727562, "step": 2116 }, { "clip_ratio/high_max": 0.006211843807250261, "clip_ratio/high_mean": 0.006211843807250261, "clip_ratio/low_mean": 0.010289846803061664, "clip_ratio/low_min": 0.010289846803061664, "clip_ratio/region_mean": 0.016501690610311925, "completions/clipped_ratio": 0.0, "completions/max_length": 186.0, "completions/max_terminated_length": 186.0, "completions/mean_length": 182.125, "completions/mean_terminated_length": 182.125, "completions/min_length": 178.0, "completions/min_terminated_length": 178.0, "entropy": 0.09395178686827421, "epoch": 0.08503032493874764, "frac_reward_zero_std": 0.0, "grad_norm": 1.891755223274231, "learning_rate": 3.587878787878788e-06, "loss": 0.0088, "num_tokens": 4771637.0, "reward": 0.7980640530586243, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9905871152877808, "reward_meter_std": 0.00902599561959505, "reward_repeat_penalty_mean": 0.8055555820465088, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.053137242794036865, "reward_total_composite_mean": 0.7980640530586243, "reward_total_composite_std": 0.05313723534345627, "reward_total_mean": 0.7980640530586243, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9905871152877808, "rewards/meter/std": 0.00902599561959505, "rewards/repeat_penalty/mean": 0.8055555820465088, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.7980640530586243, "rewards/total_composite/std": 0.05313723534345627, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0019904375076294, "sampling/importance_sampling_ratio/min": 0.27762100100517273, "sampling/sampling_logp_difference/max": 1.2814984321594238, "sampling/sampling_logp_difference/mean": 0.018866930156946182, "step": 2117 }, { "clip_ratio/high_max": 0.02530877012759447, "clip_ratio/high_mean": 0.02530877012759447, "clip_ratio/low_mean": 0.00718731596134603, "clip_ratio/low_min": 0.00718731596134603, "clip_ratio/region_mean": 0.0324960860889405, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 297.75, "completions/mean_terminated_length": 297.75, "completions/min_length": 279.0, "completions/min_terminated_length": 279.0, "entropy": 0.424762025475502, "epoch": 0.0850704904205326, "frac_reward_zero_std": 0.0, "grad_norm": 1.509689211845398, "learning_rate": 3.584848484848485e-06, "loss": -0.0196, "num_tokens": 4775851.0, "reward": 0.7875304222106934, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99297034740448, "reward_meter_std": 0.005611373111605644, "reward_repeat_penalty_mean": 0.8921875357627869, "reward_repeat_penalty_std": 0.1552397757768631, "reward_std": 0.13774563372135162, "reward_total_composite_mean": 0.7875304222106934, "reward_total_composite_std": 0.13774561882019043, "reward_total_mean": 0.7875304222106934, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99297034740448, "rewards/meter/std": 0.005611373111605644, "rewards/repeat_penalty/mean": 0.8921875357627869, "rewards/repeat_penalty/std": 0.1552397757768631, "rewards/total_composite/mean": 0.7875304222106934, "rewards/total_composite/std": 0.13774561882019043, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.00875985622406, "sampling/importance_sampling_ratio/min": 0.20513120293617249, "sampling/sampling_logp_difference/max": 1.5841054916381836, "sampling/sampling_logp_difference/mean": 0.04715384542942047, "step": 2118 }, { "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.004545454401522875, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 55.625, "completions/mean_terminated_length": 55.625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.08348313788883388, "epoch": 0.08511065590231755, "frac_reward_zero_std": 0.0, "grad_norm": 3.0825119018554688, "learning_rate": 3.5818181818181817e-06, "loss": -0.004, "num_tokens": 4777544.0, "reward": 0.9949027895927429, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949027895927429, "reward_meter_std": 0.00035745359491556883, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00035746910725720227, "reward_total_composite_mean": 0.9949027895927429, "reward_total_composite_std": 0.00035745359491556883, "reward_total_mean": 0.9949027895927429, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949027895927429, "rewards/meter/std": 0.00035745359491556883, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949027895927429, "rewards/total_composite/std": 0.00035745359491556883, "sampling/importance_sampling_ratio/max": 1.3233060836791992, "sampling/importance_sampling_ratio/mean": 0.9991430640220642, "sampling/importance_sampling_ratio/min": 0.515595555305481, "sampling/sampling_logp_difference/max": 0.6624326705932617, "sampling/sampling_logp_difference/mean": 0.01183076947927475, "step": 2119 }, { "clip_ratio/high_max": 0.020273972768336535, "clip_ratio/high_mean": 0.020273972768336535, "clip_ratio/low_mean": 0.015343417064286768, "clip_ratio/low_min": 0.015343417064286768, "clip_ratio/region_mean": 0.0356173898326233, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 73.625, "completions/mean_terminated_length": 73.625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.3053772822022438, "epoch": 0.0851508213841025, "frac_reward_zero_std": 0.0, "grad_norm": 3.3534421920776367, "learning_rate": 3.578787878787879e-06, "loss": 0.0047, "num_tokens": 4779365.0, "reward": 0.9964693784713745, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9964693784713745, "reward_meter_std": 0.002390751615166664, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002390736248344183, "reward_total_composite_mean": 0.9964693784713745, "reward_total_composite_std": 0.002390751615166664, "reward_total_mean": 0.9964693784713745, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9964693784713745, "rewards/meter/std": 0.002390751615166664, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964693784713745, "rewards/total_composite/std": 0.002390751615166664, "sampling/importance_sampling_ratio/max": 1.3884204626083374, "sampling/importance_sampling_ratio/mean": 0.9964186549186707, "sampling/importance_sampling_ratio/min": 0.24798838794231415, "sampling/sampling_logp_difference/max": 1.3943734169006348, "sampling/sampling_logp_difference/mean": 0.03976722061634064, "step": 2120 }, { "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.004546957788988948, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 55.25, "completions/mean_terminated_length": 55.25, "completions/min_length": 52.0, "completions/min_terminated_length": 52.0, "entropy": 0.053725593723356724, "epoch": 0.08519098686588746, "frac_reward_zero_std": 0.0, "grad_norm": 4.196082592010498, "learning_rate": 3.575757575757576e-06, "loss": -0.0199, "num_tokens": 4781079.0, "reward": 0.994369626045227, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994369626045227, "reward_meter_std": 0.0011783665977418423, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0011783792870119214, "reward_total_composite_mean": 0.994369626045227, "reward_total_composite_std": 0.0011783665977418423, "reward_total_mean": 0.994369626045227, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994369626045227, "rewards/meter/std": 0.0011783665977418423, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994369626045227, "rewards/total_composite/std": 0.0011783665977418423, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0018255710601807, "sampling/importance_sampling_ratio/min": 0.7468781471252441, "sampling/sampling_logp_difference/max": 0.9451119899749756, "sampling/sampling_logp_difference/mean": 0.00968268234282732, "step": 2121 }, { "clip_ratio/high_max": 0.01417637791018933, "clip_ratio/high_mean": 0.01417637791018933, "clip_ratio/low_mean": 0.009774382109753788, "clip_ratio/low_min": 0.009774382109753788, "clip_ratio/region_mean": 0.023950760019943118, "completions/clipped_ratio": 0.0, "completions/max_length": 429.0, "completions/max_terminated_length": 429.0, "completions/mean_length": 410.125, "completions/mean_terminated_length": 410.125, "completions/min_length": 383.0, "completions/min_terminated_length": 383.0, "entropy": 0.21682057529687881, "epoch": 0.08523115234767241, "frac_reward_zero_std": 0.0, "grad_norm": 1.6565197706222534, "learning_rate": 3.5727272727272734e-06, "loss": -0.0042, "num_tokens": 4786520.0, "reward": 0.5489572286605835, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.671875, "reward_count_adherence_std": 0.0289318785071373, "reward_meter_mean": 0.9933267831802368, "reward_meter_std": 0.007835081778466702, "reward_repeat_penalty_mean": 0.82259202003479, "reward_repeat_penalty_std": 0.032492659986019135, "reward_std": 0.031959328800439835, "reward_total_composite_mean": 0.5489572286605835, "reward_total_composite_std": 0.031959354877471924, "reward_total_mean": 0.5489572286605835, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.671875, "rewards/count_adherence/std": 0.0289318785071373, "rewards/meter/mean": 0.9933267831802368, "rewards/meter/std": 0.007835081778466702, "rewards/repeat_penalty/mean": 0.82259202003479, "rewards/repeat_penalty/std": 0.032492659986019135, "rewards/total_composite/mean": 0.5489572286605835, "rewards/total_composite/std": 0.031959354877471924, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0041075944900513, "sampling/importance_sampling_ratio/min": 0.014572354033589363, "sampling/sampling_logp_difference/max": 4.228629112243652, "sampling/sampling_logp_difference/mean": 0.030476173385977745, "step": 2122 }, { "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018939394503831863, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.04304229188710451, "epoch": 0.08527131782945736, "frac_reward_zero_std": 0.0, "grad_norm": 0.5491177439689636, "learning_rate": 3.5696969696969703e-06, "loss": 0.0001, "num_tokens": 4788264.0, "reward": 0.999085009098053, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999085009098053, "reward_meter_std": 1.2789209904440213e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.2776405128533952e-05, "reward_total_composite_mean": 0.999085009098053, "reward_total_composite_std": 1.2789209904440213e-05, "reward_total_mean": 0.999085009098053, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999085009098053, "rewards/meter/std": 1.2789209904440213e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999085009098053, "rewards/total_composite/std": 1.2789209904440213e-05, "sampling/importance_sampling_ratio/max": 1.2900317907333374, "sampling/importance_sampling_ratio/mean": 1.0031498670578003, "sampling/importance_sampling_ratio/min": 0.8502018451690674, "sampling/sampling_logp_difference/max": 0.254666805267334, "sampling/sampling_logp_difference/mean": 0.004210304468870163, "step": 2123 }, { "clip_ratio/high_max": 0.0031250000465661287, "clip_ratio/high_mean": 0.0031250000465661287, "clip_ratio/low_mean": 0.017462147399783134, "clip_ratio/low_min": 0.017462147399783134, "clip_ratio/region_mean": 0.020587147446349263, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 82.125, "completions/mean_terminated_length": 82.125, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.26322684064507484, "epoch": 0.08531148331124232, "frac_reward_zero_std": 0.0, "grad_norm": 5.764961242675781, "learning_rate": 3.566666666666667e-06, "loss": 0.0436, "num_tokens": 4790353.0, "reward": 0.4380970001220703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.4380970001220703, "reward_meter_std": 0.3100660443305969, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3100660443305969, "reward_total_composite_mean": 0.4380970001220703, "reward_total_composite_std": 0.3100660443305969, "reward_total_mean": 0.4380970001220703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.4380970001220703, "rewards/meter/std": 0.3100660443305969, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4380970001220703, "rewards/total_composite/std": 0.3100660443305969, "sampling/importance_sampling_ratio/max": 1.603717565536499, "sampling/importance_sampling_ratio/mean": 1.0140416622161865, "sampling/importance_sampling_ratio/min": 0.27450987696647644, "sampling/sampling_logp_difference/max": 1.2927680015563965, "sampling/sampling_logp_difference/mean": 0.030523981899023056, "step": 2124 }, { "clip_ratio/high_max": 0.0072432131273671985, "clip_ratio/high_mean": 0.0072432131273671985, "clip_ratio/low_mean": 0.0007861634949222207, "clip_ratio/low_min": 0.0007861634949222207, "clip_ratio/region_mean": 0.00802937662228942, "completions/clipped_ratio": 0.0, "completions/max_length": 159.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 156.125, "completions/mean_terminated_length": 156.125, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.03148621576838195, "epoch": 0.08535164879302727, "frac_reward_zero_std": 0.0, "grad_norm": 1.9213858842849731, "learning_rate": 3.563636363636364e-06, "loss": 0.0064, "num_tokens": 4793170.0, "reward": 0.7687686681747437, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9884169101715088, "reward_meter_std": 0.011561714112758636, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008992421440780163, "reward_total_composite_mean": 0.7687686681747437, "reward_total_composite_std": 0.008992438204586506, "reward_total_mean": 0.7687686681747437, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9884169101715088, "rewards/meter/std": 0.011561714112758636, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7687686681747437, "rewards/total_composite/std": 0.008992438204586506, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000543475151062, "sampling/importance_sampling_ratio/min": 0.17520882189273834, "sampling/sampling_logp_difference/max": 1.741776704788208, "sampling/sampling_logp_difference/mean": 0.008227693848311901, "step": 2125 }, { "clip_ratio/high_max": 0.018605169840157032, "clip_ratio/high_mean": 0.018605169840157032, "clip_ratio/low_mean": 0.018939394503831863, "clip_ratio/low_min": 0.018939394503831863, "clip_ratio/region_mean": 0.037544564343988895, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.125, "completions/mean_terminated_length": 33.125, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.14932595752179623, "epoch": 0.08539181427481222, "frac_reward_zero_std": 0.0, "grad_norm": 6.276111125946045, "learning_rate": 3.560606060606061e-06, "loss": -0.0063, "num_tokens": 4794531.0, "reward": 0.9689929485321045, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9689929485321045, "reward_meter_std": 0.0023667889181524515, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0023667817004024982, "reward_total_composite_mean": 0.9689929485321045, "reward_total_composite_std": 0.0023667889181524515, "reward_total_mean": 0.9689929485321045, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9689929485321045, "rewards/meter/std": 0.0023667889181524515, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9689929485321045, "rewards/total_composite/std": 0.0023667889181524515, "sampling/importance_sampling_ratio/max": 1.4770609140396118, "sampling/importance_sampling_ratio/mean": 1.0059295892715454, "sampling/importance_sampling_ratio/min": 0.4668806791305542, "sampling/sampling_logp_difference/max": 0.7616815567016602, "sampling/sampling_logp_difference/mean": 0.0230213962495327, "step": 2126 }, { "clip_ratio/high_max": 0.0069444444961845875, "clip_ratio/high_mean": 0.0069444444961845875, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0069444444961845875, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 54.125, "completions/mean_terminated_length": 54.125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.049557043705135584, "epoch": 0.08543197975659718, "frac_reward_zero_std": 0.0, "grad_norm": 3.2434964179992676, "learning_rate": 3.557575757575758e-06, "loss": 0.0053, "num_tokens": 4796260.0, "reward": 0.7054018974304199, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7054018974304199, "reward_meter_std": 0.08442744612693787, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.08442744612693787, "reward_total_composite_mean": 0.7054018974304199, "reward_total_composite_std": 0.08442744612693787, "reward_total_mean": 0.7054018974304199, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7054018974304199, "rewards/meter/std": 0.08442744612693787, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7054018974304199, "rewards/total_composite/std": 0.08442744612693787, "sampling/importance_sampling_ratio/max": 1.4377515316009521, "sampling/importance_sampling_ratio/mean": 1.0031954050064087, "sampling/importance_sampling_ratio/min": 0.5838686227798462, "sampling/sampling_logp_difference/max": 0.5380792617797852, "sampling/sampling_logp_difference/mean": 0.009495479986071587, "step": 2127 }, { "clip_ratio/high_max": 0.02282354235649109, "clip_ratio/high_mean": 0.02282354235649109, "clip_ratio/low_mean": 0.009215709753334522, "clip_ratio/low_min": 0.009215709753334522, "clip_ratio/region_mean": 0.03203925210982561, "completions/clipped_ratio": 0.0, "completions/max_length": 268.0, "completions/max_terminated_length": 268.0, "completions/mean_length": 261.125, "completions/mean_terminated_length": 261.125, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.43808240443468094, "epoch": 0.08547214523838213, "frac_reward_zero_std": 0.0, "grad_norm": 2.1206531524658203, "learning_rate": 3.554545454545455e-06, "loss": -0.0058, "num_tokens": 4799957.0, "reward": 0.8880020976066589, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9929981231689453, "reward_meter_std": 0.0028132039587944746, "reward_repeat_penalty_mean": 0.8942307829856873, "reward_repeat_penalty_std": 0.05723259970545769, "reward_std": 0.05752360448241234, "reward_total_composite_mean": 0.8880020976066589, "reward_total_composite_std": 0.05752360075712204, "reward_total_mean": 0.8880020976066589, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9929981231689453, "rewards/meter/std": 0.0028132039587944746, "rewards/repeat_penalty/mean": 0.8942307829856873, "rewards/repeat_penalty/std": 0.05723259970545769, "rewards/total_composite/mean": 0.8880020976066589, "rewards/total_composite/std": 0.05752360075712204, "sampling/importance_sampling_ratio/max": 1.9891395568847656, "sampling/importance_sampling_ratio/mean": 1.0112462043762207, "sampling/importance_sampling_ratio/min": 0.38044506311416626, "sampling/sampling_logp_difference/max": 0.9664134979248047, "sampling/sampling_logp_difference/mean": 0.04205208271741867, "step": 2128 }, { "clip_ratio/high_max": 0.02008880244102329, "clip_ratio/high_mean": 0.02008880244102329, "clip_ratio/low_mean": 0.01029538200236857, "clip_ratio/low_min": 0.01029538200236857, "clip_ratio/region_mean": 0.03038418444339186, "completions/clipped_ratio": 0.0, "completions/max_length": 281.0, "completions/max_terminated_length": 281.0, "completions/mean_length": 264.625, "completions/mean_terminated_length": 264.625, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "entropy": 0.35909293219447136, "epoch": 0.08551231072016709, "frac_reward_zero_std": 0.0, "grad_norm": 2.319450855255127, "learning_rate": 3.551515151515152e-06, "loss": -0.0205, "num_tokens": 4803610.0, "reward": 0.8904905915260315, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9950700998306274, "reward_meter_std": 0.0019416395807638764, "reward_repeat_penalty_mean": 0.8949176073074341, "reward_repeat_penalty_std": 0.04042290523648262, "reward_std": 0.039885781705379486, "reward_total_composite_mean": 0.8904905915260315, "reward_total_composite_std": 0.039885781705379486, "reward_total_mean": 0.8904905915260315, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9950700998306274, "rewards/meter/std": 0.0019416395807638764, "rewards/repeat_penalty/mean": 0.8949176073074341, "rewards/repeat_penalty/std": 0.04042290523648262, "rewards/total_composite/mean": 0.8904905915260315, "rewards/total_composite/std": 0.039885781705379486, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0098717212677002, "sampling/importance_sampling_ratio/min": 0.01911584474146366, "sampling/sampling_logp_difference/max": 3.957237720489502, "sampling/sampling_logp_difference/mean": 0.040030911564826965, "step": 2129 }, { "clip_ratio/high_max": 0.01770029927138239, "clip_ratio/high_mean": 0.01770029927138239, "clip_ratio/low_mean": 0.017461657291278243, "clip_ratio/low_min": 0.017461657291278243, "clip_ratio/region_mean": 0.035161956562660635, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 70.875, "completions/mean_terminated_length": 70.875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.20211379043757915, "epoch": 0.08555247620195204, "frac_reward_zero_std": 0.0, "grad_norm": 4.817035675048828, "learning_rate": 3.548484848484849e-06, "loss": 0.0131, "num_tokens": 4805553.0, "reward": 0.9955647587776184, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955647587776184, "reward_meter_std": 0.0008866624557413161, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008866624557413161, "reward_total_composite_mean": 0.9955647587776184, "reward_total_composite_std": 0.0008866624557413161, "reward_total_mean": 0.9955647587776184, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955647587776184, "rewards/meter/std": 0.0008866624557413161, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955647587776184, "rewards/total_composite/std": 0.0008866624557413161, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006104826927185, "sampling/importance_sampling_ratio/min": 0.3325047492980957, "sampling/sampling_logp_difference/max": 1.1011011600494385, "sampling/sampling_logp_difference/mean": 0.037295129150152206, "step": 2130 }, { "clip_ratio/high_max": 0.014961764682084322, "clip_ratio/high_mean": 0.014961764682084322, "clip_ratio/low_mean": 0.015936652896925807, "clip_ratio/low_min": 0.015936652896925807, "clip_ratio/region_mean": 0.03089841757901013, "completions/clipped_ratio": 0.0, "completions/max_length": 146.0, "completions/max_terminated_length": 146.0, "completions/mean_length": 142.0, "completions/mean_terminated_length": 142.0, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "entropy": 0.28764577955007553, "epoch": 0.085592641683737, "frac_reward_zero_std": 0.0, "grad_norm": 2.6760542392730713, "learning_rate": 3.5454545454545458e-06, "loss": -0.0121, "num_tokens": 4808185.0, "reward": 0.9969066381454468, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969066381454468, "reward_meter_std": 0.0009874114766716957, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009874175302684307, "reward_total_composite_mean": 0.9969066381454468, "reward_total_composite_std": 0.0009874114766716957, "reward_total_mean": 0.9969066381454468, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969066381454468, "rewards/meter/std": 0.0009874114766716957, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969066381454468, "rewards/total_composite/std": 0.0009874114766716957, "sampling/importance_sampling_ratio/max": 1.913059115409851, "sampling/importance_sampling_ratio/mean": 1.0037314891815186, "sampling/importance_sampling_ratio/min": 0.2744320034980774, "sampling/sampling_logp_difference/max": 1.2930517196655273, "sampling/sampling_logp_difference/mean": 0.03481653332710266, "step": 2131 }, { "clip_ratio/high_max": 0.026603603502735496, "clip_ratio/high_mean": 0.026603603502735496, "clip_ratio/low_mean": 0.004672897048294544, "clip_ratio/low_min": 0.004672897048294544, "clip_ratio/region_mean": 0.03127650055103004, "completions/clipped_ratio": 0.0, "completions/max_length": 112.0, "completions/max_terminated_length": 112.0, "completions/mean_length": 107.875, "completions/mean_terminated_length": 107.875, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.17346674110740423, "epoch": 0.08563280716552195, "frac_reward_zero_std": 0.0, "grad_norm": 3.024273633956909, "learning_rate": 3.5424242424242426e-06, "loss": 0.0002, "num_tokens": 4810360.0, "reward": 0.9690351486206055, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939501285552979, "reward_meter_std": 0.003658889327198267, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06948734074831009, "reward_total_composite_mean": 0.9690351486206055, "reward_total_composite_std": 0.06948735564947128, "reward_total_mean": 0.9690351486206055, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939501285552979, "rewards/meter/std": 0.003658889327198267, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9690351486206055, "rewards/total_composite/std": 0.06948735564947128, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0037057399749756, "sampling/importance_sampling_ratio/min": 0.2526026964187622, "sampling/sampling_logp_difference/max": 1.3759374618530273, "sampling/sampling_logp_difference/mean": 0.03451121971011162, "step": 2132 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0030303029343485832, "clip_ratio/low_min": 0.0030303029343485832, "clip_ratio/region_mean": 0.0030303029343485832, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 165.5, "completions/mean_terminated_length": 165.5, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.03646064782515168, "epoch": 0.0856729726473069, "frac_reward_zero_std": 0.0, "grad_norm": 0.6826779246330261, "learning_rate": 3.53939393939394e-06, "loss": 0.001, "num_tokens": 4813068.0, "reward": 0.7910317182540894, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999198317527771, "reward_meter_std": 3.6679448385257274e-05, "reward_repeat_penalty_mean": 0.7916666865348816, "reward_repeat_penalty_std": 0.03928370773792267, "reward_std": 0.03924502432346344, "reward_total_composite_mean": 0.7910317182540894, "reward_total_composite_std": 0.039245039224624634, "reward_total_mean": 0.7910317182540894, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999198317527771, "rewards/meter/std": 3.6679448385257274e-05, "rewards/repeat_penalty/mean": 0.7916666865348816, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.7910317182540894, "rewards/total_composite/std": 0.039245039224624634, "sampling/importance_sampling_ratio/max": 1.5342899560928345, "sampling/importance_sampling_ratio/mean": 1.0013805627822876, "sampling/importance_sampling_ratio/min": 0.19314932823181152, "sampling/sampling_logp_difference/max": 1.644291639328003, "sampling/sampling_logp_difference/mean": 0.008064919151365757, "step": 2133 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/region_mean": 0.004629629664123058, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.037100087851285934, "epoch": 0.08571313812909186, "frac_reward_zero_std": 0.0, "grad_norm": 0.44613105058670044, "learning_rate": 3.5363636363636367e-06, "loss": 0.0003, "num_tokens": 4814740.0, "reward": 0.7352948188781738, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7352948188781738, "reward_meter_std": 3.051780367968604e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.0517016057274304e-05, "reward_total_composite_mean": 0.7352948188781738, "reward_total_composite_std": 3.051780367968604e-05, "reward_total_mean": 0.7352948188781738, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7352948188781738, "rewards/meter/std": 3.051780367968604e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7352948188781738, "rewards/total_composite/std": 3.051780367968604e-05, "sampling/importance_sampling_ratio/max": 1.9774119853973389, "sampling/importance_sampling_ratio/mean": 1.0043625831604004, "sampling/importance_sampling_ratio/min": 0.7626789808273315, "sampling/sampling_logp_difference/max": 0.6817889213562012, "sampling/sampling_logp_difference/mean": 0.005636686459183693, "step": 2134 }, { "clip_ratio/high_max": 0.014304422307759523, "clip_ratio/high_mean": 0.014304422307759523, "clip_ratio/low_mean": 0.016251615015789866, "clip_ratio/low_min": 0.016251615015789866, "clip_ratio/region_mean": 0.03055603732354939, "completions/clipped_ratio": 0.0, "completions/max_length": 240.0, "completions/max_terminated_length": 240.0, "completions/mean_length": 229.375, "completions/mean_terminated_length": 229.375, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.4052262455224991, "epoch": 0.08575330361087681, "frac_reward_zero_std": 0.0, "grad_norm": 4.601060390472412, "learning_rate": 3.5333333333333335e-06, "loss": 0.0222, "num_tokens": 4818135.0, "reward": 0.9174596071243286, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.971871018409729, "reward_meter_std": 0.06841260194778442, "reward_repeat_penalty_mean": 0.9431818723678589, "reward_repeat_penalty_std": 0.047049909830093384, "reward_std": 0.08795087039470673, "reward_total_composite_mean": 0.9174596071243286, "reward_total_composite_std": 0.08795089274644852, "reward_total_mean": 0.9174596071243286, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.971871018409729, "rewards/meter/std": 0.06841260194778442, "rewards/repeat_penalty/mean": 0.9431818723678589, "rewards/repeat_penalty/std": 0.047049909830093384, "rewards/total_composite/mean": 0.9174596071243286, "rewards/total_composite/std": 0.08795089274644852, "sampling/importance_sampling_ratio/max": 1.8775804042816162, "sampling/importance_sampling_ratio/mean": 1.0070955753326416, "sampling/importance_sampling_ratio/min": 0.18075646460056305, "sampling/sampling_logp_difference/max": 1.7106046676635742, "sampling/sampling_logp_difference/mean": 0.041487276554107666, "step": 2135 }, { "clip_ratio/high_max": 0.0015015150420367718, "clip_ratio/high_mean": 0.0015015150420367718, "clip_ratio/low_mean": 0.004527199547737837, "clip_ratio/low_min": 0.004527199547737837, "clip_ratio/region_mean": 0.006028714589774609, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 165.875, "completions/mean_terminated_length": 165.875, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.06440420588478446, "epoch": 0.08579346909266176, "frac_reward_zero_std": 0.0, "grad_norm": 1.9712090492248535, "learning_rate": 3.5303030303030304e-06, "loss": -0.0019, "num_tokens": 4820878.0, "reward": 0.8604106903076172, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991864562034607, "reward_meter_std": 2.615616722323466e-05, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.07856741547584534, "reward_std": 0.07850649952888489, "reward_total_composite_mean": 0.8604106903076172, "reward_total_composite_std": 0.07850649952888489, "reward_total_mean": 0.8604106903076172, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991864562034607, "rewards/meter/std": 2.615616722323466e-05, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8604106903076172, "rewards/total_composite/std": 0.07850649952888489, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0031864643096924, "sampling/importance_sampling_ratio/min": 0.5129733085632324, "sampling/sampling_logp_difference/max": 0.7467837333679199, "sampling/sampling_logp_difference/mean": 0.009452839381992817, "step": 2136 }, { "clip_ratio/high_max": 0.0025862068869173527, "clip_ratio/high_mean": 0.0025862068869173527, "clip_ratio/low_mean": 0.006010864395648241, "clip_ratio/low_min": 0.006010864395648241, "clip_ratio/region_mean": 0.008597071282565594, "completions/clipped_ratio": 0.0, "completions/max_length": 148.0, "completions/max_terminated_length": 148.0, "completions/mean_length": 145.625, "completions/mean_terminated_length": 145.625, "completions/min_length": 145.0, "completions/min_terminated_length": 145.0, "entropy": 0.04937975388020277, "epoch": 0.08583363457444672, "frac_reward_zero_std": 0.0, "grad_norm": 1.4629026651382446, "learning_rate": 3.5272727272727276e-06, "loss": -0.0007, "num_tokens": 4823547.0, "reward": 0.7541667222976685, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9696428775787354, "reward_meter_std": 0.0028882811311632395, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0022464566864073277, "reward_total_composite_mean": 0.7541667222976685, "reward_total_composite_std": 0.0022464508656412363, "reward_total_mean": 0.7541667222976685, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9696428775787354, "rewards/meter/std": 0.0028882811311632395, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7541667222976685, "rewards/total_composite/std": 0.0022464508656412363, "sampling/importance_sampling_ratio/max": 1.7878315448760986, "sampling/importance_sampling_ratio/mean": 1.0017269849777222, "sampling/importance_sampling_ratio/min": 0.26717090606689453, "sampling/sampling_logp_difference/max": 1.31986665725708, "sampling/sampling_logp_difference/mean": 0.011477367021143436, "step": 2137 }, { "clip_ratio/high_max": 0.02402331749908626, "clip_ratio/high_mean": 0.02402331749908626, "clip_ratio/low_mean": 0.01095851231366396, "clip_ratio/low_min": 0.01095851231366396, "clip_ratio/region_mean": 0.03498182981275022, "completions/clipped_ratio": 0.0, "completions/max_length": 233.0, "completions/max_terminated_length": 233.0, "completions/mean_length": 228.75, "completions/mean_terminated_length": 228.75, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "entropy": 0.40121684968471527, "epoch": 0.08587380005623167, "frac_reward_zero_std": 0.0, "grad_norm": 2.5688741207122803, "learning_rate": 3.5242424242424244e-06, "loss": 0.0015, "num_tokens": 4826897.0, "reward": 0.9418025016784668, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9738757610321045, "reward_meter_std": 0.055281247943639755, "reward_repeat_penalty_mean": 0.9659091234207153, "reward_repeat_penalty_std": 0.047049909830093384, "reward_std": 0.08390163630247116, "reward_total_composite_mean": 0.9418025016784668, "reward_total_composite_std": 0.08390164375305176, "reward_total_mean": 0.9418025016784668, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9738757610321045, "rewards/meter/std": 0.055281247943639755, "rewards/repeat_penalty/mean": 0.9659091234207153, "rewards/repeat_penalty/std": 0.047049909830093384, "rewards/total_composite/mean": 0.9418025016784668, "rewards/total_composite/std": 0.08390164375305176, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0137181282043457, "sampling/importance_sampling_ratio/min": 0.20458358526229858, "sampling/sampling_logp_difference/max": 1.5867786407470703, "sampling/sampling_logp_difference/mean": 0.04527745395898819, "step": 2138 }, { "clip_ratio/high_max": 0.002314814832061529, "clip_ratio/high_mean": 0.002314814832061529, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.002314814832061529, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.020970646291971207, "epoch": 0.08591396553801663, "frac_reward_zero_std": 0.0, "grad_norm": 1.6966145038604736, "learning_rate": 3.5212121212121213e-06, "loss": 0.0018, "num_tokens": 4828577.0, "reward": 0.7483930587768555, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7483930587768555, "reward_meter_std": 0.02422301471233368, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.024223024025559425, "reward_total_composite_mean": 0.7483930587768555, "reward_total_composite_std": 0.02422301471233368, "reward_total_mean": 0.7483930587768555, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7483930587768555, "rewards/meter/std": 0.02422301471233368, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7483930587768555, "rewards/total_composite/std": 0.02422301471233368, "sampling/importance_sampling_ratio/max": 1.6018749475479126, "sampling/importance_sampling_ratio/mean": 1.0017740726470947, "sampling/importance_sampling_ratio/min": 0.6358863711357117, "sampling/sampling_logp_difference/max": 0.4711747169494629, "sampling/sampling_logp_difference/mean": 0.005149809177964926, "step": 2139 }, { "clip_ratio/high_max": 0.02094770153053105, "clip_ratio/high_mean": 0.02094770153053105, "clip_ratio/low_mean": 0.004959081998094916, "clip_ratio/low_min": 0.004959081998094916, "clip_ratio/region_mean": 0.025906783528625965, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.375, "completions/mean_terminated_length": 76.375, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.3370419256389141, "epoch": 0.08595413101980158, "frac_reward_zero_std": 0.0, "grad_norm": 3.241760015487671, "learning_rate": 3.5181818181818185e-06, "loss": -0.0033, "num_tokens": 4830436.0, "reward": 0.9935569763183594, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9935569763183594, "reward_meter_std": 0.004046422429382801, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004046411253511906, "reward_total_composite_mean": 0.9935569763183594, "reward_total_composite_std": 0.004046422429382801, "reward_total_mean": 0.9935569763183594, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9935569763183594, "rewards/meter/std": 0.004046422429382801, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9935569763183594, "rewards/total_composite/std": 0.004046422429382801, "sampling/importance_sampling_ratio/max": 1.7292462587356567, "sampling/importance_sampling_ratio/mean": 1.008273720741272, "sampling/importance_sampling_ratio/min": 0.2990627586841583, "sampling/sampling_logp_difference/max": 1.207101821899414, "sampling/sampling_logp_difference/mean": 0.03363337367773056, "step": 2140 }, { "clip_ratio/high_max": 0.004464285913854837, "clip_ratio/high_mean": 0.004464285913854837, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.004464285913854837, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 55.875, "completions/mean_terminated_length": 55.875, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.0797986383549869, "epoch": 0.08599429650158653, "frac_reward_zero_std": 0.0, "grad_norm": 4.896651268005371, "learning_rate": 3.5151515151515154e-06, "loss": 0.0014, "num_tokens": 4832051.0, "reward": 0.9941670894622803, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9941670894622803, "reward_meter_std": 0.002000106731429696, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0020000922959297895, "reward_total_composite_mean": 0.9941670894622803, "reward_total_composite_std": 0.002000106731429696, "reward_total_mean": 0.9941670894622803, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9941670894622803, "rewards/meter/std": 0.002000106731429696, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9941670894622803, "rewards/total_composite/std": 0.002000106731429696, "sampling/importance_sampling_ratio/max": 1.2935843467712402, "sampling/importance_sampling_ratio/mean": 1.0028151273727417, "sampling/importance_sampling_ratio/min": 0.4471496343612671, "sampling/sampling_logp_difference/max": 0.8048620223999023, "sampling/sampling_logp_difference/mean": 0.011897793039679527, "step": 2141 }, { "clip_ratio/high_max": 0.027495570946484804, "clip_ratio/high_mean": 0.027495570946484804, "clip_ratio/low_mean": 0.005864104256033897, "clip_ratio/low_min": 0.005864104256033897, "clip_ratio/region_mean": 0.0333596752025187, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 428.0, "completions/mean_terminated_length": 428.0, "completions/min_length": 411.0, "completions/min_terminated_length": 411.0, "entropy": 0.3651280999183655, "epoch": 0.08603446198337149, "frac_reward_zero_std": 0.0, "grad_norm": 1.6217327117919922, "learning_rate": 3.512121212121212e-06, "loss": 0.013, "num_tokens": 4837339.0, "reward": 0.6134706139564514, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6953125, "reward_count_adherence_std": 0.022097086533904076, "reward_meter_mean": 0.9931625127792358, "reward_meter_std": 0.0038301812019199133, "reward_repeat_penalty_mean": 0.8884575366973877, "reward_repeat_penalty_std": 0.061322376132011414, "reward_std": 0.04515310376882553, "reward_total_composite_mean": 0.6134706139564514, "reward_total_composite_std": 0.04515310749411583, "reward_total_mean": 0.6134706139564514, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6953125, "rewards/count_adherence/std": 0.022097086533904076, "rewards/meter/mean": 0.9931625127792358, "rewards/meter/std": 0.0038301812019199133, "rewards/repeat_penalty/mean": 0.8884575366973877, "rewards/repeat_penalty/std": 0.061322376132011414, "rewards/total_composite/mean": 0.6134706139564514, "rewards/total_composite/std": 0.04515310749411583, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0085182189941406, "sampling/importance_sampling_ratio/min": 0.05357210338115692, "sampling/sampling_logp_difference/max": 2.926726818084717, "sampling/sampling_logp_difference/mean": 0.042900450527668, "step": 2142 }, { "clip_ratio/high_max": 0.024497864302247763, "clip_ratio/high_mean": 0.024497864302247763, "clip_ratio/low_mean": 0.040365297347307205, "clip_ratio/low_min": 0.040365297347307205, "clip_ratio/region_mean": 0.06486316164955497, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 74.75, "completions/mean_terminated_length": 74.75, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.6021371968090534, "epoch": 0.08607462746515644, "frac_reward_zero_std": 0.0, "grad_norm": 5.659821510314941, "learning_rate": 3.509090909090909e-06, "loss": 0.0064, "num_tokens": 4839153.0, "reward": 0.9972976446151733, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972976446151733, "reward_meter_std": 0.0019415906863287091, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0019415918504819274, "reward_total_composite_mean": 0.9972976446151733, "reward_total_composite_std": 0.0019415906863287091, "reward_total_mean": 0.9972976446151733, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972976446151733, "rewards/meter/std": 0.0019415906863287091, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972976446151733, "rewards/total_composite/std": 0.0019415906863287091, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0229203701019287, "sampling/importance_sampling_ratio/min": 0.4317888617515564, "sampling/sampling_logp_difference/max": 1.5144741535186768, "sampling/sampling_logp_difference/mean": 0.06912153214216232, "step": 2143 }, { "clip_ratio/high_max": 0.025277130538597703, "clip_ratio/high_mean": 0.025277130538597703, "clip_ratio/low_mean": 0.021914413664489985, "clip_ratio/low_min": 0.021914413664489985, "clip_ratio/region_mean": 0.04719154420308769, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 74.75, "completions/mean_terminated_length": 74.75, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.48777300119400024, "epoch": 0.0861147929469414, "frac_reward_zero_std": 0.0, "grad_norm": 5.293501853942871, "learning_rate": 3.5060606060606063e-06, "loss": -0.0009, "num_tokens": 4841063.0, "reward": 0.9942959547042847, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942959547042847, "reward_meter_std": 0.0059119779616594315, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005911969114094973, "reward_total_composite_mean": 0.9942959547042847, "reward_total_composite_std": 0.0059119779616594315, "reward_total_mean": 0.9942959547042847, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942959547042847, "rewards/meter/std": 0.0059119779616594315, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942959547042847, "rewards/total_composite/std": 0.0059119779616594315, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0201735496520996, "sampling/importance_sampling_ratio/min": 0.09109176695346832, "sampling/sampling_logp_difference/max": 2.395887851715088, "sampling/sampling_logp_difference/mean": 0.06987594068050385, "step": 2144 }, { "clip_ratio/high_max": 0.01278735650703311, "clip_ratio/high_mean": 0.01278735650703311, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.01278735650703311, "completions/clipped_ratio": 0.0, "completions/max_length": 30.0, "completions/max_terminated_length": 30.0, "completions/mean_length": 29.25, "completions/mean_terminated_length": 29.25, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.09351515863090754, "epoch": 0.08615495842872635, "frac_reward_zero_std": 0.0, "grad_norm": 7.162275314331055, "learning_rate": 3.503030303030303e-06, "loss": 0.0132, "num_tokens": 4842409.0, "reward": 0.9901993274688721, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9901993274688721, "reward_meter_std": 0.004359308164566755, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0043593174777925014, "reward_total_composite_mean": 0.9901993274688721, "reward_total_composite_std": 0.004359308164566755, "reward_total_mean": 0.9901993274688721, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9901993274688721, "rewards/meter/std": 0.004359308164566755, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9901993274688721, "rewards/total_composite/std": 0.004359308164566755, "sampling/importance_sampling_ratio/max": 1.2712383270263672, "sampling/importance_sampling_ratio/mean": 0.9988601207733154, "sampling/importance_sampling_ratio/min": 0.6107580661773682, "sampling/sampling_logp_difference/max": 0.4930543899536133, "sampling/sampling_logp_difference/mean": 0.01579299196600914, "step": 2145 }, { "clip_ratio/high_max": 0.02845356403850019, "clip_ratio/high_mean": 0.02845356403850019, "clip_ratio/low_mean": 0.008386766072362661, "clip_ratio/low_min": 0.008386766072362661, "clip_ratio/region_mean": 0.03684033011086285, "completions/clipped_ratio": 0.0, "completions/max_length": 488.0, "completions/max_terminated_length": 488.0, "completions/mean_length": 198.75, "completions/mean_terminated_length": 198.75, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 1.2716873697936535, "epoch": 0.0861951239105113, "frac_reward_zero_std": 0.0, "grad_norm": 5.25761604309082, "learning_rate": 3.5e-06, "loss": 0.5018, "num_tokens": 4845311.0, "reward": 0.7369348406791687, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.26726123690605164, "reward_meter_mean": 0.8164126873016357, "reward_meter_std": 0.32973745465278625, "reward_repeat_penalty_mean": 0.9434523582458496, "reward_repeat_penalty_std": 0.0783882662653923, "reward_std": 0.32373327016830444, "reward_total_composite_mean": 0.7369348406791687, "reward_total_composite_std": 0.32373327016830444, "reward_total_mean": 0.7369348406791687, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.26726123690605164, "rewards/meter/mean": 0.8164126873016357, "rewards/meter/std": 0.32973745465278625, "rewards/repeat_penalty/mean": 0.9434523582458496, "rewards/repeat_penalty/std": 0.0783882662653923, "rewards/total_composite/mean": 0.7369348406791687, "rewards/total_composite/std": 0.32373327016830444, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.013696312904358, "sampling/importance_sampling_ratio/min": 0.20175102353096008, "sampling/sampling_logp_difference/max": 1.6007208824157715, "sampling/sampling_logp_difference/mean": 0.09025763720273972, "step": 2146 }, { "clip_ratio/high_max": 0.029611698118969798, "clip_ratio/high_mean": 0.029611698118969798, "clip_ratio/low_mean": 0.00345633109100163, "clip_ratio/low_min": 0.00345633109100163, "clip_ratio/region_mean": 0.03306802920997143, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 73.25, "completions/mean_terminated_length": 73.25, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.41511600464582443, "epoch": 0.08623528939229626, "frac_reward_zero_std": 0.0, "grad_norm": 7.887787342071533, "learning_rate": 3.496969696969697e-06, "loss": -0.0056, "num_tokens": 4847185.0, "reward": 0.8275381922721863, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8275381922721863, "reward_meter_std": 0.3271496891975403, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3271496891975403, "reward_total_composite_mean": 0.8275381922721863, "reward_total_composite_std": 0.3271496891975403, "reward_total_mean": 0.8275381922721863, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8275381922721863, "rewards/meter/std": 0.3271496891975403, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8275381922721863, "rewards/total_composite/std": 0.3271496891975403, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0108306407928467, "sampling/importance_sampling_ratio/min": 0.17972400784492493, "sampling/sampling_logp_difference/max": 1.7163329124450684, "sampling/sampling_logp_difference/mean": 0.0584087148308754, "step": 2147 }, { "clip_ratio/high_max": 0.012931034434586763, "clip_ratio/high_mean": 0.012931034434586763, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/region_mean": 0.017241379246115685, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.0592546658590436, "epoch": 0.08627545487408121, "frac_reward_zero_std": 0.0, "grad_norm": 6.055057525634766, "learning_rate": 3.493939393939394e-06, "loss": -0.0013, "num_tokens": 4848601.0, "reward": 0.9926831722259521, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9926831722259521, "reward_meter_std": 0.0014186145272105932, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0014186184853315353, "reward_total_composite_mean": 0.9926831722259521, "reward_total_composite_std": 0.0014186145272105932, "reward_total_mean": 0.9926831722259521, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9926831722259521, "rewards/meter/std": 0.0014186145272105932, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926831722259521, "rewards/total_composite/std": 0.0014186145272105932, "sampling/importance_sampling_ratio/max": 1.2208077907562256, "sampling/importance_sampling_ratio/mean": 0.9984536170959473, "sampling/importance_sampling_ratio/min": 0.4162471890449524, "sampling/sampling_logp_difference/max": 0.8764760494232178, "sampling/sampling_logp_difference/mean": 0.018664227798581123, "step": 2148 }, { "clip_ratio/high_max": 0.02710459241643548, "clip_ratio/high_mean": 0.02710459241643548, "clip_ratio/low_mean": 0.021443208097480237, "clip_ratio/low_min": 0.021443208097480237, "clip_ratio/region_mean": 0.04854780051391572, "completions/clipped_ratio": 0.0, "completions/max_length": 196.0, "completions/max_terminated_length": 196.0, "completions/mean_length": 186.375, "completions/mean_terminated_length": 186.375, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.5572481229901314, "epoch": 0.08631562035586617, "frac_reward_zero_std": 0.0, "grad_norm": 3.5529558658599854, "learning_rate": 3.4909090909090913e-06, "loss": 0.012, "num_tokens": 4851588.0, "reward": 0.8901457786560059, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9706813097000122, "reward_meter_std": 0.06584999710321426, "reward_repeat_penalty_mean": 0.9166666269302368, "reward_repeat_penalty_std": 0.07856741547584534, "reward_std": 0.10105939954519272, "reward_total_composite_mean": 0.8901457786560059, "reward_total_composite_std": 0.10105938464403152, "reward_total_mean": 0.8901457786560059, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9706813097000122, "rewards/meter/std": 0.06584999710321426, "rewards/repeat_penalty/mean": 0.9166666269302368, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8901457786560059, "rewards/total_composite/std": 0.10105938464403152, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0136346817016602, "sampling/importance_sampling_ratio/min": 0.19374443590641022, "sampling/sampling_logp_difference/max": 1.6412153244018555, "sampling/sampling_logp_difference/mean": 0.05906498804688454, "step": 2149 }, { "clip_ratio/high_max": 0.009757570689544082, "clip_ratio/high_mean": 0.009757570689544082, "clip_ratio/low_mean": 0.006343001441564411, "clip_ratio/low_min": 0.006343001441564411, "clip_ratio/region_mean": 0.016100572131108493, "completions/clipped_ratio": 0.0, "completions/max_length": 182.0, "completions/max_terminated_length": 182.0, "completions/mean_length": 179.25, "completions/mean_terminated_length": 179.25, "completions/min_length": 174.0, "completions/min_terminated_length": 174.0, "entropy": 0.2794443480670452, "epoch": 0.08635578583765112, "frac_reward_zero_std": 0.0, "grad_norm": 2.0675745010375977, "learning_rate": 3.4878787878787885e-06, "loss": 0.0075, "num_tokens": 4854574.0, "reward": 0.9418361783027649, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972676634788513, "reward_meter_std": 0.0014413135359063745, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.08399210125207901, "reward_std": 0.0834418535232544, "reward_total_composite_mean": 0.9418361783027649, "reward_total_composite_std": 0.08344186097383499, "reward_total_mean": 0.9418361783027649, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972676634788513, "rewards/meter/std": 0.0014413135359063745, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.9418361783027649, "rewards/total_composite/std": 0.08344186097383499, "sampling/importance_sampling_ratio/max": 1.8844407796859741, "sampling/importance_sampling_ratio/mean": 1.0084565877914429, "sampling/importance_sampling_ratio/min": 0.3118909001350403, "sampling/sampling_logp_difference/max": 1.1651017665863037, "sampling/sampling_logp_difference/mean": 0.02511134371161461, "step": 2150 }, { "epoch": 0.08635578583765112, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 377.61538461538464, "eval_completions/max_terminated_length": 377.61538461538464, "eval_completions/mean_length": 208.15384615384616, "eval_completions/mean_terminated_length": 208.15384615384616, "eval_completions/min_length": 61.23076923076923, "eval_completions/min_terminated_length": 61.23076923076923, "eval_entropy": 0.20996456879835862, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 4854574.0, "eval_reward": 0.5750110378632178, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9253275165191064, "eval_reward_count_adherence_std": 0.10161341182314433, "eval_reward_meter_mean": 0.7046629740641668, "eval_reward_meter_std": 0.41940173277488124, "eval_reward_repeat_penalty_mean": 0.843710142832536, "eval_reward_repeat_penalty_std": 0.15254983076682457, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5750110378632178, "eval_reward_total_composite_std": 0.38267656473013073, "eval_reward_total_mean": 0.5750110378632178, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9253275165191064, "eval_rewards/count_adherence/std": 0.10161341182314433, "eval_rewards/meter/mean": 0.7046629740641668, "eval_rewards/meter/std": 0.41940173277488124, "eval_rewards/repeat_penalty/mean": 0.843710142832536, "eval_rewards/repeat_penalty/std": 0.15254983076682457, "eval_rewards/total_composite/mean": 0.5750110378632178, "eval_rewards/total_composite/std": 0.38267656473013073, "eval_runtime": 71.0335, "eval_samples_per_second": 1.464, "eval_sampling/importance_sampling_ratio/max": 1.5623481090252216, "eval_sampling/importance_sampling_ratio/mean": 1.0052896371254554, "eval_sampling/importance_sampling_ratio/min": 0.3407066028851729, "eval_sampling/sampling_logp_difference/max": 1.116743463736314, "eval_sampling/sampling_logp_difference/mean": 0.02115864851153814, "eval_steps_per_second": 0.183, "step": 2150 }, { "clip_ratio/high_max": 0.023466601967811584, "clip_ratio/high_mean": 0.023466601967811584, "clip_ratio/low_mean": 0.06512659136205912, "clip_ratio/low_min": 0.06512659136205912, "clip_ratio/region_mean": 0.0885931933298707, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 44.0, "completions/mean_length": 101.625, "completions/mean_terminated_length": 43.000003814697266, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.9084643498063087, "epoch": 0.08639595131943607, "frac_reward_zero_std": 0.0, "grad_norm": 3.3067500591278076, "learning_rate": 3.4848484848484854e-06, "loss": -0.0636, "num_tokens": 4856155.0, "reward": 0.4681951105594635, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.1767766922712326, "reward_meter_mean": 0.4682069420814514, "reward_meter_std": 0.35492372512817383, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35494154691696167, "reward_total_composite_mean": 0.4681951105594635, "reward_total_composite_std": 0.35494157671928406, "reward_total_mean": 0.4681951105594635, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.1767766922712326, "rewards/meter/mean": 0.4682069420814514, "rewards/meter/std": 0.35492372512817383, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.4681951105594635, "rewards/total_composite/std": 0.35494157671928406, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0368061065673828, "sampling/importance_sampling_ratio/min": 0.07572062313556671, "sampling/sampling_logp_difference/max": 2.580704689025879, "sampling/sampling_logp_difference/mean": 0.11403389275074005, "step": 2151 }, { "clip_ratio/high_max": 0.0109593840315938, "clip_ratio/high_mean": 0.0109593840315938, "clip_ratio/low_mean": 0.010588002507574856, "clip_ratio/low_min": 0.010588002507574856, "clip_ratio/region_mean": 0.021547386539168656, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 247.5, "completions/mean_terminated_length": 247.5, "completions/min_length": 238.0, "completions/min_terminated_length": 238.0, "entropy": 0.1678062528371811, "epoch": 0.08643611680122103, "frac_reward_zero_std": 0.0, "grad_norm": 6.426714897155762, "learning_rate": 3.481818181818182e-06, "loss": 0.0533, "num_tokens": 4859871.0, "reward": 0.8597774505615234, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9642857313156128, "reward_count_adherence_std": 0.06613000482320786, "reward_meter_mean": 0.9982793927192688, "reward_meter_std": 0.00048778695054352283, "reward_repeat_penalty_mean": 0.8910256624221802, "reward_repeat_penalty_std": 0.06309091299772263, "reward_std": 0.10398054867982864, "reward_total_composite_mean": 0.8597774505615234, "reward_total_composite_std": 0.10398055613040924, "reward_total_mean": 0.8597774505615234, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9642857313156128, "rewards/count_adherence/std": 0.06613000482320786, "rewards/meter/mean": 0.9982793927192688, "rewards/meter/std": 0.00048778695054352283, "rewards/repeat_penalty/mean": 0.8910256624221802, "rewards/repeat_penalty/std": 0.06309091299772263, "rewards/total_composite/mean": 0.8597774505615234, "rewards/total_composite/std": 0.10398055613040924, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0031156539916992, "sampling/importance_sampling_ratio/min": 0.03819035738706589, "sampling/sampling_logp_difference/max": 3.265172243118286, "sampling/sampling_logp_difference/mean": 0.028371067717671394, "step": 2152 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.007575757801532745, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.07956603448837996, "epoch": 0.08647628228300598, "frac_reward_zero_std": 0.0, "grad_norm": 0.24814273416996002, "learning_rate": 3.4787878787878795e-06, "loss": 0.0, "num_tokens": 4861871.0, "reward": 0.9990808367729187, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990808367729187, "reward_meter_std": 1.3548809874919243e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.3551140909839887e-05, "reward_total_composite_mean": 0.9990808367729187, "reward_total_composite_std": 1.3548809874919243e-05, "reward_total_mean": 0.9990808367729187, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990808367729187, "rewards/meter/std": 1.3548809874919243e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990808367729187, "rewards/total_composite/std": 1.3548809874919243e-05, "sampling/importance_sampling_ratio/max": 1.5080136060714722, "sampling/importance_sampling_ratio/mean": 1.0004948377609253, "sampling/importance_sampling_ratio/min": 0.2651733458042145, "sampling/sampling_logp_difference/max": 1.327371597290039, "sampling/sampling_logp_difference/mean": 0.011720891110599041, "step": 2153 }, { "clip_ratio/high_max": 0.022406263276934624, "clip_ratio/high_mean": 0.022406263276934624, "clip_ratio/low_mean": 0.011893743183463812, "clip_ratio/low_min": 0.011893743183463812, "clip_ratio/region_mean": 0.034300006460398436, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 72.75, "completions/mean_terminated_length": 72.75, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.23252216633409262, "epoch": 0.08651644776479094, "frac_reward_zero_std": 0.0, "grad_norm": 3.136981248855591, "learning_rate": 3.4757575757575763e-06, "loss": 0.0058, "num_tokens": 4863701.0, "reward": 0.9944970607757568, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9944970607757568, "reward_meter_std": 0.004257019143551588, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004257019609212875, "reward_total_composite_mean": 0.9944970607757568, "reward_total_composite_std": 0.004257019143551588, "reward_total_mean": 0.9944970607757568, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9944970607757568, "rewards/meter/std": 0.004257019143551588, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944970607757568, "rewards/total_composite/std": 0.004257019143551588, "sampling/importance_sampling_ratio/max": 1.6701947450637817, "sampling/importance_sampling_ratio/mean": 1.0077894926071167, "sampling/importance_sampling_ratio/min": 0.2449745386838913, "sampling/sampling_logp_difference/max": 1.4066009521484375, "sampling/sampling_logp_difference/mean": 0.03498362377285957, "step": 2154 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.005681818351149559, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.07073036208748817, "epoch": 0.08655661324657589, "frac_reward_zero_std": 0.0, "grad_norm": 0.40986379981040955, "learning_rate": 3.472727272727273e-06, "loss": -0.0003, "num_tokens": 4865461.0, "reward": 0.9990643262863159, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990643262863159, "reward_meter_std": 2.489090729795862e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.4881244826246984e-05, "reward_total_composite_mean": 0.9990643262863159, "reward_total_composite_std": 2.489090729795862e-05, "reward_total_mean": 0.9990643262863159, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990643262863159, "rewards/meter/std": 2.489090729795862e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990643262863159, "rewards/total_composite/std": 2.489090729795862e-05, "sampling/importance_sampling_ratio/max": 1.318129062652588, "sampling/importance_sampling_ratio/mean": 1.0042916536331177, "sampling/importance_sampling_ratio/min": 0.8046210408210754, "sampling/sampling_logp_difference/max": 0.27621328830718994, "sampling/sampling_logp_difference/mean": 0.005097586195915937, "step": 2155 }, { "clip_ratio/high_max": 0.009525301051326096, "clip_ratio/high_mean": 0.009525301051326096, "clip_ratio/low_mean": 0.0032144944416359067, "clip_ratio/low_min": 0.0032144944416359067, "clip_ratio/region_mean": 0.012739795492962003, "completions/clipped_ratio": 0.0, "completions/max_length": 119.0, "completions/max_terminated_length": 119.0, "completions/mean_length": 117.625, "completions/mean_terminated_length": 117.625, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.13127075601369143, "epoch": 0.08659677872836084, "frac_reward_zero_std": 0.0, "grad_norm": 1.6892346143722534, "learning_rate": 3.46969696969697e-06, "loss": -0.0024, "num_tokens": 4867682.0, "reward": 0.9429587125778198, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963239431381226, "reward_meter_std": 0.0006225909455679357, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07379541546106339, "reward_total_composite_mean": 0.9429587125778198, "reward_total_composite_std": 0.07379542291164398, "reward_total_mean": 0.9429587125778198, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963239431381226, "rewards/meter/std": 0.0006225909455679357, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9429587125778198, "rewards/total_composite/std": 0.07379542291164398, "sampling/importance_sampling_ratio/max": 1.3534270524978638, "sampling/importance_sampling_ratio/mean": 1.0034620761871338, "sampling/importance_sampling_ratio/min": 0.22397485375404358, "sampling/sampling_logp_difference/max": 1.4962215423583984, "sampling/sampling_logp_difference/mean": 0.020759226754307747, "step": 2156 }, { "clip_ratio/high_max": 0.00889376224949956, "clip_ratio/high_mean": 0.00889376224949956, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/region_mean": 0.013279727194458246, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 56.625, "completions/mean_terminated_length": 56.625, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.054770099464803934, "epoch": 0.0866369442101458, "frac_reward_zero_std": 0.0, "grad_norm": 1.6599547863006592, "learning_rate": 3.4666666666666672e-06, "loss": 0.0158, "num_tokens": 4869399.0, "reward": 0.725243330001831, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.725243330001831, "reward_meter_std": 0.03141331672668457, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03141331672668457, "reward_total_composite_mean": 0.725243330001831, "reward_total_composite_std": 0.03141331672668457, "reward_total_mean": 0.725243330001831, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.725243330001831, "rewards/meter/std": 0.03141331672668457, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.725243330001831, "rewards/total_composite/std": 0.03141331672668457, "sampling/importance_sampling_ratio/max": 1.4941579103469849, "sampling/importance_sampling_ratio/mean": 1.0009208917617798, "sampling/importance_sampling_ratio/min": 0.5721346139907837, "sampling/sampling_logp_difference/max": 0.5583809614181519, "sampling/sampling_logp_difference/mean": 0.01150722336024046, "step": 2157 }, { "clip_ratio/high_max": 0.005681818351149559, "clip_ratio/high_mean": 0.005681818351149559, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.009469697251915932, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.07476874813437462, "epoch": 0.08667710969193075, "frac_reward_zero_std": 0.0, "grad_norm": 0.1878003627061844, "learning_rate": 3.463636363636364e-06, "loss": 0.0002, "num_tokens": 4871311.0, "reward": 0.9990821480751038, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990821480751038, "reward_meter_std": 1.2787881132680923e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.2793108908226714e-05, "reward_total_composite_mean": 0.9990821480751038, "reward_total_composite_std": 1.2787881132680923e-05, "reward_total_mean": 0.9990821480751038, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990821480751038, "rewards/meter/std": 1.2787881132680923e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990821480751038, "rewards/total_composite/std": 1.2787881132680923e-05, "sampling/importance_sampling_ratio/max": 1.3374110460281372, "sampling/importance_sampling_ratio/mean": 1.0013294219970703, "sampling/importance_sampling_ratio/min": 0.33294302225112915, "sampling/sampling_logp_difference/max": 1.0997838973999023, "sampling/sampling_logp_difference/mean": 0.008319023996591568, "step": 2158 }, { "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0022321429569274187, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 56.125, "completions/mean_terminated_length": 56.125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.093318160623312, "epoch": 0.0867172751737157, "frac_reward_zero_std": 0.0, "grad_norm": 3.910881280899048, "learning_rate": 3.460606060606061e-06, "loss": 0.0063, "num_tokens": 4873104.0, "reward": 0.9947191476821899, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9947191476821899, "reward_meter_std": 0.0006506324862129986, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006506353965960443, "reward_total_composite_mean": 0.9947191476821899, "reward_total_composite_std": 0.0006506324862129986, "reward_total_mean": 0.9947191476821899, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9947191476821899, "rewards/meter/std": 0.0006506324862129986, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9947191476821899, "rewards/total_composite/std": 0.0006506324862129986, "sampling/importance_sampling_ratio/max": 1.261582612991333, "sampling/importance_sampling_ratio/mean": 1.002916693687439, "sampling/importance_sampling_ratio/min": 0.6279736161231995, "sampling/sampling_logp_difference/max": 0.4652571678161621, "sampling/sampling_logp_difference/mean": 0.009572095237672329, "step": 2159 }, { "clip_ratio/high_max": 0.01438528229482472, "clip_ratio/high_mean": 0.01438528229482472, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/region_mean": 0.01869562710635364, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 87.0, "completions/mean_terminated_length": 87.0, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.11874265689402819, "epoch": 0.08675744065550066, "frac_reward_zero_std": 0.0, "grad_norm": 1.6953576803207397, "learning_rate": 3.4575757575757577e-06, "loss": 0.0003, "num_tokens": 4875040.0, "reward": 0.9958928227424622, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958928227424622, "reward_meter_std": 0.000379539153072983, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000379539153072983, "reward_total_composite_mean": 0.9958928227424622, "reward_total_composite_std": 0.000379539153072983, "reward_total_mean": 0.9958928227424622, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958928227424622, "rewards/meter/std": 0.000379539153072983, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9958928227424622, "rewards/total_composite/std": 0.000379539153072983, "sampling/importance_sampling_ratio/max": 1.3476742506027222, "sampling/importance_sampling_ratio/mean": 1.0018218755722046, "sampling/importance_sampling_ratio/min": 0.38639914989471436, "sampling/sampling_logp_difference/max": 0.9508843421936035, "sampling/sampling_logp_difference/mean": 0.018249215558171272, "step": 2160 }, { "clip_ratio/high_max": 0.006696428870782256, "clip_ratio/high_mean": 0.006696428870782256, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006696428870782256, "completions/clipped_ratio": 0.0, "completions/max_length": 57.0, "completions/max_terminated_length": 57.0, "completions/mean_length": 56.125, "completions/mean_terminated_length": 56.125, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.06367568951100111, "epoch": 0.08679760613728561, "frac_reward_zero_std": 0.0, "grad_norm": 2.359445810317993, "learning_rate": 3.454545454545455e-06, "loss": 0.0029, "num_tokens": 4876737.0, "reward": 0.9948000907897949, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948000907897949, "reward_meter_std": 0.0004447593819350004, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00044475641334429383, "reward_total_composite_mean": 0.9948000907897949, "reward_total_composite_std": 0.0004447593819350004, "reward_total_mean": 0.9948000907897949, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948000907897949, "rewards/meter/std": 0.0004447593819350004, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948000907897949, "rewards/total_composite/std": 0.0004447593819350004, "sampling/importance_sampling_ratio/max": 1.2896134853363037, "sampling/importance_sampling_ratio/mean": 1.0003728866577148, "sampling/importance_sampling_ratio/min": 0.42699676752090454, "sampling/sampling_logp_difference/max": 0.8509788513183594, "sampling/sampling_logp_difference/mean": 0.007567924913018942, "step": 2161 }, { "clip_ratio/high_max": 0.0022727272007614374, "clip_ratio/high_mean": 0.0022727272007614374, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/region_mean": 0.004823747556656599, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 53.5, "completions/mean_terminated_length": 53.5, "completions/min_length": 49.0, "completions/min_terminated_length": 49.0, "entropy": 0.08292486914433539, "epoch": 0.08683777161907057, "frac_reward_zero_std": 0.0, "grad_norm": 4.7188615798950195, "learning_rate": 3.451515151515152e-06, "loss": -0.0273, "num_tokens": 4878413.0, "reward": 0.709714949131012, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7199991941452026, "reward_meter_std": 0.19119207561016083, "reward_repeat_penalty_mean": 0.9583333730697632, "reward_repeat_penalty_std": 0.117851123213768, "reward_std": 0.22028043866157532, "reward_total_composite_mean": 0.709714949131012, "reward_total_composite_std": 0.2202804535627365, "reward_total_mean": 0.709714949131012, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7199991941452026, "rewards/meter/std": 0.19119207561016083, "rewards/repeat_penalty/mean": 0.9583333730697632, "rewards/repeat_penalty/std": 0.117851123213768, "rewards/total_composite/mean": 0.709714949131012, "rewards/total_composite/std": 0.2202804535627365, "sampling/importance_sampling_ratio/max": 1.4815089702606201, "sampling/importance_sampling_ratio/mean": 0.9992796778678894, "sampling/importance_sampling_ratio/min": 0.21796490252017975, "sampling/sampling_logp_difference/max": 1.523421287536621, "sampling/sampling_logp_difference/mean": 0.014019476249814034, "step": 2162 }, { "clip_ratio/high_max": 0.011834641918540001, "clip_ratio/high_mean": 0.011834641918540001, "clip_ratio/low_mean": 0.00980698096100241, "clip_ratio/low_min": 0.00980698096100241, "clip_ratio/region_mean": 0.02164162287954241, "completions/clipped_ratio": 0.0, "completions/max_length": 272.0, "completions/max_terminated_length": 272.0, "completions/mean_length": 254.625, "completions/mean_terminated_length": 254.625, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.32685666531324387, "epoch": 0.08687793710085552, "frac_reward_zero_std": 0.0, "grad_norm": 2.0742194652557373, "learning_rate": 3.4484848484848486e-06, "loss": 0.0136, "num_tokens": 4882602.0, "reward": 0.8155835270881653, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978955984115601, "reward_meter_std": 0.0005918731912970543, "reward_repeat_penalty_mean": 0.9340659379959106, "reward_repeat_penalty_std": 0.062163230031728745, "reward_std": 0.054202865809202194, "reward_total_composite_mean": 0.8155835270881653, "reward_total_composite_std": 0.05420287698507309, "reward_total_mean": 0.8155835270881653, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978955984115601, "rewards/meter/std": 0.0005918731912970543, "rewards/repeat_penalty/mean": 0.9340659379959106, "rewards/repeat_penalty/std": 0.062163230031728745, "rewards/total_composite/mean": 0.8155835270881653, "rewards/total_composite/std": 0.05420287698507309, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0046485662460327, "sampling/importance_sampling_ratio/min": 1.0665352334626732e-07, "sampling/sampling_logp_difference/max": 16.053680419921875, "sampling/sampling_logp_difference/mean": 0.04149610176682472, "step": 2163 }, { "clip_ratio/high_max": 0.0055147059028968215, "clip_ratio/high_mean": 0.0055147059028968215, "clip_ratio/low_mean": 0.00378874852322042, "clip_ratio/low_min": 0.00378874852322042, "clip_ratio/region_mean": 0.009303454426117241, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.75, "completions/mean_terminated_length": 67.75, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.14008523616939783, "epoch": 0.08691810258264047, "frac_reward_zero_std": 0.0, "grad_norm": 3.962695837020874, "learning_rate": 3.445454545454546e-06, "loss": -0.007, "num_tokens": 4884392.0, "reward": 0.9326764345169067, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9326764345169067, "reward_meter_std": 0.024432551115751266, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.024432554841041565, "reward_total_composite_mean": 0.9326764345169067, "reward_total_composite_std": 0.024432551115751266, "reward_total_mean": 0.9326764345169067, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9326764345169067, "rewards/meter/std": 0.024432551115751266, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9326764345169067, "rewards/total_composite/std": 0.024432551115751266, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0046216249465942, "sampling/importance_sampling_ratio/min": 0.3380107283592224, "sampling/sampling_logp_difference/max": 1.0846776962280273, "sampling/sampling_logp_difference/mean": 0.021882515400648117, "step": 2164 }, { "clip_ratio/high_max": 0.007464349502697587, "clip_ratio/high_mean": 0.007464349502697587, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007464349502697587, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.25, "completions/mean_terminated_length": 66.25, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.10618306044489145, "epoch": 0.08695826806442543, "frac_reward_zero_std": 0.0, "grad_norm": 3.4584054946899414, "learning_rate": 3.4424242424242427e-06, "loss": -0.0061, "num_tokens": 4886154.0, "reward": 0.9357738494873047, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9357738494873047, "reward_meter_std": 0.17855486273765564, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.17855484783649445, "reward_total_composite_mean": 0.9357738494873047, "reward_total_composite_std": 0.17855486273765564, "reward_total_mean": 0.9357738494873047, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9357738494873047, "rewards/meter/std": 0.17855486273765564, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9357738494873047, "rewards/total_composite/std": 0.17855486273765564, "sampling/importance_sampling_ratio/max": 1.5800968408584595, "sampling/importance_sampling_ratio/mean": 1.00041925907135, "sampling/importance_sampling_ratio/min": 0.0004117942589800805, "sampling/sampling_logp_difference/max": 7.794986724853516, "sampling/sampling_logp_difference/mean": 0.030108362436294556, "step": 2165 }, { "clip_ratio/high_max": 0.008708675275556743, "clip_ratio/high_mean": 0.008708675275556743, "clip_ratio/low_mean": 0.005332982516847551, "clip_ratio/low_min": 0.005332982516847551, "clip_ratio/region_mean": 0.014041657792404294, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 116.125, "completions/mean_terminated_length": 116.125, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.13098172564059496, "epoch": 0.08699843354621038, "frac_reward_zero_std": 0.0, "grad_norm": 2.468413829803467, "learning_rate": 3.4393939393939395e-06, "loss": 0.0044, "num_tokens": 4888587.0, "reward": 0.9249789118766785, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996168851852417, "reward_meter_std": 0.0008722886559553444, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.07558359950780869, "reward_total_composite_mean": 0.9249789118766785, "reward_total_composite_std": 0.07558359950780869, "reward_total_mean": 0.9249789118766785, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996168851852417, "rewards/meter/std": 0.0008722886559553444, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9249789118766785, "rewards/total_composite/std": 0.07558359950780869, "sampling/importance_sampling_ratio/max": 1.6350321769714355, "sampling/importance_sampling_ratio/mean": 1.0041885375976562, "sampling/importance_sampling_ratio/min": 0.41527071595191956, "sampling/sampling_logp_difference/max": 0.8788247108459473, "sampling/sampling_logp_difference/mean": 0.019451651722192764, "step": 2166 }, { "clip_ratio/high_max": 0.024240323924459517, "clip_ratio/high_mean": 0.024240323924459517, "clip_ratio/low_mean": 0.008205128135159612, "clip_ratio/low_min": 0.008205128135159612, "clip_ratio/region_mean": 0.03244545205961913, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 73.0, "completions/mean_terminated_length": 73.0, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.2611173652112484, "epoch": 0.08703859902799534, "frac_reward_zero_std": 0.0, "grad_norm": 7.104674816131592, "learning_rate": 3.4363636363636364e-06, "loss": 0.0303, "num_tokens": 4890379.0, "reward": 0.9878561496734619, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9878561496734619, "reward_meter_std": 0.010053827427327633, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010053832083940506, "reward_total_composite_mean": 0.9878561496734619, "reward_total_composite_std": 0.010053827427327633, "reward_total_mean": 0.9878561496734619, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9878561496734619, "rewards/meter/std": 0.010053827427327633, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9878561496734619, "rewards/total_composite/std": 0.010053827427327633, "sampling/importance_sampling_ratio/max": 1.616224765777588, "sampling/importance_sampling_ratio/mean": 1.0013821125030518, "sampling/importance_sampling_ratio/min": 0.2027558535337448, "sampling/sampling_logp_difference/max": 1.5957527160644531, "sampling/sampling_logp_difference/mean": 0.037653032690286636, "step": 2167 }, { "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0035714285913854837, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.051043334417045116, "epoch": 0.08707876450978029, "frac_reward_zero_std": 0.0, "grad_norm": 0.1801041215658188, "learning_rate": 3.4333333333333336e-06, "loss": -0.0006, "num_tokens": 4891883.0, "reward": 0.9979922771453857, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979922771453857, "reward_meter_std": 1.613328822713811e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.613025233382359e-05, "reward_total_composite_mean": 0.9979922771453857, "reward_total_composite_std": 1.613328822713811e-05, "reward_total_mean": 0.9979922771453857, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979922771453857, "rewards/meter/std": 1.613328822713811e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979922771453857, "rewards/total_composite/std": 1.613328822713811e-05, "sampling/importance_sampling_ratio/max": 1.0980141162872314, "sampling/importance_sampling_ratio/mean": 1.0019093751907349, "sampling/importance_sampling_ratio/min": 0.7093939781188965, "sampling/sampling_logp_difference/max": 0.34334421157836914, "sampling/sampling_logp_difference/mean": 0.005629383493214846, "step": 2168 }, { "clip_ratio/high_max": 0.022110585356131196, "clip_ratio/high_mean": 0.022110585356131196, "clip_ratio/low_mean": 0.006560013280250132, "clip_ratio/low_min": 0.006560013280250132, "clip_ratio/region_mean": 0.028670598636381328, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 346.625, "completions/mean_terminated_length": 346.625, "completions/min_length": 336.0, "completions/min_terminated_length": 336.0, "entropy": 0.32041275314986706, "epoch": 0.08711892999156524, "frac_reward_zero_std": 0.0, "grad_norm": 1.7812119722366333, "learning_rate": 3.4303030303030305e-06, "loss": 0.0195, "num_tokens": 4896528.0, "reward": 0.6071589589118958, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.701923131942749, "reward_count_adherence_std": 0.027196412906050682, "reward_meter_mean": 0.9944499731063843, "reward_meter_std": 0.0038915579207241535, "reward_repeat_penalty_mean": 0.8722910284996033, "reward_repeat_penalty_std": 0.08643024414777756, "reward_std": 0.045305285602808, "reward_total_composite_mean": 0.6071589589118958, "reward_total_composite_std": 0.0453052781522274, "reward_total_mean": 0.6071589589118958, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.701923131942749, "rewards/count_adherence/std": 0.027196412906050682, "rewards/meter/mean": 0.9944499731063843, "rewards/meter/std": 0.0038915579207241535, "rewards/repeat_penalty/mean": 0.8722910284996033, "rewards/repeat_penalty/std": 0.08643024414777756, "rewards/total_composite/mean": 0.6071589589118958, "rewards/total_composite/std": 0.0453052781522274, "sampling/importance_sampling_ratio/max": 1.909399151802063, "sampling/importance_sampling_ratio/mean": 1.0060648918151855, "sampling/importance_sampling_ratio/min": 0.004339110571891069, "sampling/sampling_logp_difference/max": 5.4400858879089355, "sampling/sampling_logp_difference/mean": 0.03724047169089317, "step": 2169 }, { "clip_ratio/high_max": 0.016413721488788724, "clip_ratio/high_mean": 0.016413721488788724, "clip_ratio/low_mean": 0.008402727777138352, "clip_ratio/low_min": 0.008402727777138352, "clip_ratio/region_mean": 0.024816449265927076, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 75.5, "completions/mean_terminated_length": 75.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.34711651876568794, "epoch": 0.0871590954733502, "frac_reward_zero_std": 0.0, "grad_norm": 2.5535831451416016, "learning_rate": 3.4272727272727273e-06, "loss": -0.0018, "num_tokens": 4898436.0, "reward": 0.9986087083816528, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986087083816528, "reward_meter_std": 0.0005744586233049631, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005744565278291702, "reward_total_composite_mean": 0.9986087083816528, "reward_total_composite_std": 0.0005744586233049631, "reward_total_mean": 0.9986087083816528, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986087083816528, "rewards/meter/std": 0.0005744586233049631, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986087083816528, "rewards/total_composite/std": 0.0005744586233049631, "sampling/importance_sampling_ratio/max": 1.619355320930481, "sampling/importance_sampling_ratio/mean": 1.0089645385742188, "sampling/importance_sampling_ratio/min": 0.3079758584499359, "sampling/sampling_logp_difference/max": 1.1777338981628418, "sampling/sampling_logp_difference/mean": 0.032794393599033356, "step": 2170 }, { "clip_ratio/high_max": 0.018138289218768477, "clip_ratio/high_mean": 0.018138289218768477, "clip_ratio/low_mean": 0.010196133749559522, "clip_ratio/low_min": 0.010196133749559522, "clip_ratio/region_mean": 0.028334422968328, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 75.625, "completions/mean_terminated_length": 75.625, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.3033796586096287, "epoch": 0.08719926095513515, "frac_reward_zero_std": 0.0, "grad_norm": 3.1598801612854004, "learning_rate": 3.4242424242424246e-06, "loss": 0.0221, "num_tokens": 4900337.0, "reward": 0.9978922009468079, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978922009468079, "reward_meter_std": 0.00160902866628021, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001609029364772141, "reward_total_composite_mean": 0.9978922009468079, "reward_total_composite_std": 0.00160902866628021, "reward_total_mean": 0.9978922009468079, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978922009468079, "rewards/meter/std": 0.00160902866628021, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978922009468079, "rewards/total_composite/std": 0.00160902866628021, "sampling/importance_sampling_ratio/max": 1.6152563095092773, "sampling/importance_sampling_ratio/mean": 1.0061036348342896, "sampling/importance_sampling_ratio/min": 0.30386632680892944, "sampling/sampling_logp_difference/max": 1.1911673545837402, "sampling/sampling_logp_difference/mean": 0.03502906486392021, "step": 2171 }, { "clip_ratio/high_max": 0.012681030901148915, "clip_ratio/high_mean": 0.012681030901148915, "clip_ratio/low_mean": 0.012229247367940843, "clip_ratio/low_min": 0.012229247367940843, "clip_ratio/region_mean": 0.02491027826908976, "completions/clipped_ratio": 0.0, "completions/max_length": 219.0, "completions/max_terminated_length": 219.0, "completions/mean_length": 216.375, "completions/mean_terminated_length": 216.375, "completions/min_length": 213.0, "completions/min_terminated_length": 213.0, "entropy": 0.3047393374145031, "epoch": 0.0872394264369201, "frac_reward_zero_std": 0.0, "grad_norm": 2.200758695602417, "learning_rate": 3.4212121212121214e-06, "loss": 0.0003, "num_tokens": 4903772.0, "reward": 0.8068255186080933, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980002641677856, "reward_meter_std": 0.00031345427851192653, "reward_repeat_penalty_mean": 0.9431818723678589, "reward_repeat_penalty_std": 0.06763852387666702, "reward_std": 0.057862233370542526, "reward_total_composite_mean": 0.8068255186080933, "reward_total_composite_std": 0.05786222964525223, "reward_total_mean": 0.8068255186080933, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980002641677856, "rewards/meter/std": 0.00031345427851192653, "rewards/repeat_penalty/mean": 0.9431818723678589, "rewards/repeat_penalty/std": 0.06763852387666702, "rewards/total_composite/mean": 0.8068255186080933, "rewards/total_composite/std": 0.05786222964525223, "sampling/importance_sampling_ratio/max": 1.921886920928955, "sampling/importance_sampling_ratio/mean": 1.0067384243011475, "sampling/importance_sampling_ratio/min": 0.33980512619018555, "sampling/sampling_logp_difference/max": 1.0793828964233398, "sampling/sampling_logp_difference/mean": 0.028192421421408653, "step": 2172 }, { "clip_ratio/high_max": 0.020063025411218405, "clip_ratio/high_mean": 0.020063025411218405, "clip_ratio/low_mean": 0.0036498295376077294, "clip_ratio/low_min": 0.0036498295376077294, "clip_ratio/region_mean": 0.023712854948826134, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.375, "completions/mean_terminated_length": 68.375, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.12862383667379618, "epoch": 0.08727959191870506, "frac_reward_zero_std": 0.0, "grad_norm": 4.465800762176514, "learning_rate": 3.4181818181818182e-06, "loss": -0.0002, "num_tokens": 4905495.0, "reward": 0.9475573301315308, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9475573301315308, "reward_meter_std": 0.03694632276892662, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.036946315318346024, "reward_total_composite_mean": 0.9475573301315308, "reward_total_composite_std": 0.03694632276892662, "reward_total_mean": 0.9475573301315308, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9475573301315308, "rewards/meter/std": 0.03694632276892662, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9475573301315308, "rewards/total_composite/std": 0.03694632276892662, "sampling/importance_sampling_ratio/max": 1.5417495965957642, "sampling/importance_sampling_ratio/mean": 1.000171184539795, "sampling/importance_sampling_ratio/min": 0.3909483253955841, "sampling/sampling_logp_difference/max": 0.9391798973083496, "sampling/sampling_logp_difference/mean": 0.02550523541867733, "step": 2173 }, { "clip_ratio/high_max": 0.0013020833721384406, "clip_ratio/high_mean": 0.0013020833721384406, "clip_ratio/low_mean": 0.008064516121521592, "clip_ratio/low_min": 0.008064516121521592, "clip_ratio/region_mean": 0.009366599493660033, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 93.375, "completions/mean_terminated_length": 93.375, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.06566301081329584, "epoch": 0.08731975740049001, "frac_reward_zero_std": 0.0, "grad_norm": 1.1670396327972412, "learning_rate": 3.415151515151515e-06, "loss": -0.0049, "num_tokens": 4907490.0, "reward": 0.9865900874137878, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9865900874137878, "reward_meter_std": 0.00459433114156127, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0045943367294967175, "reward_total_composite_mean": 0.9865900874137878, "reward_total_composite_std": 0.00459433114156127, "reward_total_mean": 0.9865900874137878, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9865900874137878, "rewards/meter/std": 0.00459433114156127, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9865900874137878, "rewards/total_composite/std": 0.00459433114156127, "sampling/importance_sampling_ratio/max": 1.247177004814148, "sampling/importance_sampling_ratio/mean": 0.999477744102478, "sampling/importance_sampling_ratio/min": 0.09968418627977371, "sampling/sampling_logp_difference/max": 2.305748224258423, "sampling/sampling_logp_difference/mean": 0.013781530782580376, "step": 2174 }, { "clip_ratio/high_max": 0.0058139534667134285, "clip_ratio/high_mean": 0.0058139534667134285, "clip_ratio/low_mean": 0.0029411765281111, "clip_ratio/low_min": 0.0029411765281111, "clip_ratio/region_mean": 0.008755129994824529, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 85.625, "completions/mean_terminated_length": 85.625, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.09059183578938246, "epoch": 0.08735992288227497, "frac_reward_zero_std": 0.0, "grad_norm": 2.544595956802368, "learning_rate": 3.4121212121212123e-06, "loss": -0.0057, "num_tokens": 4909423.0, "reward": 0.9955015778541565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955015778541565, "reward_meter_std": 0.00046232930617406964, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00046233335160650313, "reward_total_composite_mean": 0.9955015778541565, "reward_total_composite_std": 0.00046232930617406964, "reward_total_mean": 0.9955015778541565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955015778541565, "rewards/meter/std": 0.00046232930617406964, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955015778541565, "rewards/total_composite/std": 0.00046232930617406964, "sampling/importance_sampling_ratio/max": 1.216043472290039, "sampling/importance_sampling_ratio/mean": 1.005947232246399, "sampling/importance_sampling_ratio/min": 0.7659775018692017, "sampling/sampling_logp_difference/max": 0.2666025161743164, "sampling/sampling_logp_difference/mean": 0.009426219388842583, "step": 2175 }, { "clip_ratio/high_max": 0.006137048127129674, "clip_ratio/high_mean": 0.006137048127129674, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006137048127129674, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 80.5, "completions/mean_terminated_length": 80.5, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.035707658622413874, "epoch": 0.08740008836405992, "frac_reward_zero_std": 0.0, "grad_norm": 2.064898729324341, "learning_rate": 3.409090909090909e-06, "loss": -0.001, "num_tokens": 4911299.0, "reward": 0.69842928647995, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7326533794403076, "reward_meter_std": 0.055831361562013626, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.10409945994615555, "reward_total_composite_mean": 0.69842928647995, "reward_total_composite_std": 0.10409945249557495, "reward_total_mean": 0.69842928647995, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7326533794403076, "rewards/meter/std": 0.055831361562013626, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.69842928647995, "rewards/total_composite/std": 0.10409945249557495, "sampling/importance_sampling_ratio/max": 1.3141322135925293, "sampling/importance_sampling_ratio/mean": 0.9991859197616577, "sampling/importance_sampling_ratio/min": 0.4615878760814667, "sampling/sampling_logp_difference/max": 0.7730828523635864, "sampling/sampling_logp_difference/mean": 0.008284451439976692, "step": 2176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.009889230597764254, "epoch": 0.08744025384584488, "frac_reward_zero_std": 0.0, "grad_norm": 0.053646668791770935, "learning_rate": 3.406060606060606e-06, "loss": -0.0033, "num_tokens": 4912899.0, "reward": 0.7176748514175415, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7176748514175415, "reward_meter_std": 0.19788797199726105, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.19788797199726105, "reward_total_composite_mean": 0.7176748514175415, "reward_total_composite_std": 0.19788797199726105, "reward_total_mean": 0.7176748514175415, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7176748514175415, "rewards/meter/std": 0.19788797199726105, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7176748514175415, "rewards/total_composite/std": 0.19788797199726105, "sampling/importance_sampling_ratio/max": 1.0113025903701782, "sampling/importance_sampling_ratio/mean": 0.999964714050293, "sampling/importance_sampling_ratio/min": 0.5506253242492676, "sampling/sampling_logp_difference/max": 0.5967006683349609, "sampling/sampling_logp_difference/mean": 0.00239483080804348, "step": 2177 }, { "clip_ratio/high_max": 0.0021929824724793434, "clip_ratio/high_mean": 0.0021929824724793434, "clip_ratio/low_mean": 0.00733574153855443, "clip_ratio/low_min": 0.00733574153855443, "clip_ratio/region_mean": 0.009528724011033773, "completions/clipped_ratio": 0.0, "completions/max_length": 171.0, "completions/max_terminated_length": 171.0, "completions/mean_length": 170.625, "completions/mean_terminated_length": 170.625, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.1335141584277153, "epoch": 0.08748041932762983, "frac_reward_zero_std": 0.0, "grad_norm": 1.710945725440979, "learning_rate": 3.4030303030303036e-06, "loss": 0.0009, "num_tokens": 4915728.0, "reward": 0.9158225059509277, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990792274475098, "reward_meter_std": 7.88669494795613e-05, "reward_repeat_penalty_mean": 0.9166666269302368, "reward_repeat_penalty_std": 0.05143444612622261, "reward_std": 0.05138415843248367, "reward_total_composite_mean": 0.9158225059509277, "reward_total_composite_std": 0.05138414725661278, "reward_total_mean": 0.9158225059509277, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990792274475098, "rewards/meter/std": 7.88669494795613e-05, "rewards/repeat_penalty/mean": 0.9166666269302368, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.9158225059509277, "rewards/total_composite/std": 0.05138414725661278, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0056817531585693, "sampling/importance_sampling_ratio/min": 0.3940596580505371, "sampling/sampling_logp_difference/max": 1.5503005981445312, "sampling/sampling_logp_difference/mean": 0.01991666667163372, "step": 2178 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.005681818351149559, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.05868239142000675, "epoch": 0.08752058480941478, "frac_reward_zero_std": 0.0, "grad_norm": 0.20853394269943237, "learning_rate": 3.4000000000000005e-06, "loss": 0.0006, "num_tokens": 4917552.0, "reward": 0.9990785121917725, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990785121917725, "reward_meter_std": 9.964550372387748e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.972212865250185e-06, "reward_total_composite_mean": 0.9990785121917725, "reward_total_composite_std": 9.964550372387748e-06, "reward_total_mean": 0.9990785121917725, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990785121917725, "rewards/meter/std": 9.964550372387748e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990785121917725, "rewards/total_composite/std": 9.964550372387748e-06, "sampling/importance_sampling_ratio/max": 1.365889072418213, "sampling/importance_sampling_ratio/mean": 1.002784252166748, "sampling/importance_sampling_ratio/min": 0.6621303558349609, "sampling/sampling_logp_difference/max": 0.41229283809661865, "sampling/sampling_logp_difference/mean": 0.0065149967558681965, "step": 2179 }, { "clip_ratio/high_max": 0.02744697011075914, "clip_ratio/high_mean": 0.02744697011075914, "clip_ratio/low_mean": 0.005902778008021414, "clip_ratio/low_min": 0.005902778008021414, "clip_ratio/region_mean": 0.03334974811878055, "completions/clipped_ratio": 0.0, "completions/max_length": 236.0, "completions/max_terminated_length": 236.0, "completions/mean_length": 219.375, "completions/mean_terminated_length": 219.375, "completions/min_length": 210.0, "completions/min_terminated_length": 210.0, "entropy": 0.3587251305580139, "epoch": 0.08756075029119974, "frac_reward_zero_std": 0.0, "grad_norm": 2.8113794326782227, "learning_rate": 3.3969696969696973e-06, "loss": -0.0137, "num_tokens": 4921059.0, "reward": 0.7102342844009399, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8746868968009949, "reward_meter_std": 0.34104546904563904, "reward_repeat_penalty_mean": 0.9318181872367859, "reward_repeat_penalty_std": 0.08058229833841324, "reward_std": 0.2835994064807892, "reward_total_composite_mean": 0.7102342844009399, "reward_total_composite_std": 0.2835994064807892, "reward_total_mean": 0.7102342844009399, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8746868968009949, "rewards/meter/std": 0.34104546904563904, "rewards/repeat_penalty/mean": 0.9318181872367859, "rewards/repeat_penalty/std": 0.08058229833841324, "rewards/total_composite/mean": 0.7102342844009399, "rewards/total_composite/std": 0.2835994064807892, "sampling/importance_sampling_ratio/max": 1.8379487991333008, "sampling/importance_sampling_ratio/mean": 1.002653956413269, "sampling/importance_sampling_ratio/min": 0.27144426107406616, "sampling/sampling_logp_difference/max": 1.3039984703063965, "sampling/sampling_logp_difference/mean": 0.033945005387067795, "step": 2180 }, { "clip_ratio/high_max": 0.0064301491947844625, "clip_ratio/high_mean": 0.0064301491947844625, "clip_ratio/low_mean": 0.0025641699321568012, "clip_ratio/low_min": 0.0025641699321568012, "clip_ratio/region_mean": 0.008994319126941264, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.25, "completions/mean_terminated_length": 97.25, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.0856907800771296, "epoch": 0.08760091577298469, "frac_reward_zero_std": 0.0, "grad_norm": 2.3878560066223145, "learning_rate": 3.3939393939393946e-06, "loss": 0.0015, "num_tokens": 4923237.0, "reward": 0.8764194250106812, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8764194250106812, "reward_meter_std": 0.24076080322265625, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.24076080322265625, "reward_total_composite_mean": 0.8764194250106812, "reward_total_composite_std": 0.24076080322265625, "reward_total_mean": 0.8764194250106812, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8764194250106812, "rewards/meter/std": 0.24076080322265625, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8764194250106812, "rewards/total_composite/std": 0.24076080322265625, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0025182962417603, "sampling/importance_sampling_ratio/min": 0.3312835097312927, "sampling/sampling_logp_difference/max": 1.104780673980713, "sampling/sampling_logp_difference/mean": 0.012246760539710522, "step": 2181 }, { "clip_ratio/high_max": 0.005906250094994903, "clip_ratio/high_mean": 0.005906250094994903, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/region_mean": 0.007859375094994903, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 127.125, "completions/mean_terminated_length": 127.125, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.07762610260397196, "epoch": 0.08764108125476965, "frac_reward_zero_std": 0.0, "grad_norm": 2.051013946533203, "learning_rate": 3.3909090909090914e-06, "loss": -0.0054, "num_tokens": 4925782.0, "reward": 0.8546895980834961, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971377849578857, "reward_meter_std": 0.0010128796566277742, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008681949693709612, "reward_total_composite_mean": 0.8546895980834961, "reward_total_composite_std": 0.0008681902545504272, "reward_total_mean": 0.8546895980834961, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971377849578857, "rewards/meter/std": 0.0010128796566277742, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8546895980834961, "rewards/total_composite/std": 0.0008681902545504272, "sampling/importance_sampling_ratio/max": 1.307826042175293, "sampling/importance_sampling_ratio/mean": 1.0007635354995728, "sampling/importance_sampling_ratio/min": 0.20016026496887207, "sampling/sampling_logp_difference/max": 1.6086368560791016, "sampling/sampling_logp_difference/mean": 0.011722107417881489, "step": 2182 }, { "clip_ratio/high_max": 0.005599473137408495, "clip_ratio/high_mean": 0.005599473137408495, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005599473137408495, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.07284869812428951, "epoch": 0.0876812467365546, "frac_reward_zero_std": 0.0, "grad_norm": 3.130744457244873, "learning_rate": 3.3878787878787882e-06, "loss": 0.0002, "num_tokens": 4927521.0, "reward": 0.9979628324508667, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979628324508667, "reward_meter_std": 0.0002601691521704197, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002601697633508593, "reward_total_composite_mean": 0.9979628324508667, "reward_total_composite_std": 0.0002601691521704197, "reward_total_mean": 0.9979628324508667, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979628324508667, "rewards/meter/std": 0.0002601691521704197, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979628324508667, "rewards/total_composite/std": 0.0002601691521704197, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.00442373752594, "sampling/importance_sampling_ratio/min": 0.3769982159137726, "sampling/sampling_logp_difference/max": 0.9755148887634277, "sampling/sampling_logp_difference/mean": 0.010011628270149231, "step": 2183 }, { "clip_ratio/high_max": 0.031617405358701944, "clip_ratio/high_mean": 0.031617405358701944, "clip_ratio/low_mean": 0.014024864183738828, "clip_ratio/low_min": 0.014024864183738828, "clip_ratio/region_mean": 0.04564226954244077, "completions/clipped_ratio": 0.0, "completions/max_length": 117.0, "completions/max_terminated_length": 117.0, "completions/mean_length": 110.625, "completions/mean_terminated_length": 110.625, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.5080476962029934, "epoch": 0.08772141221833955, "frac_reward_zero_std": 0.0, "grad_norm": 4.508577346801758, "learning_rate": 3.384848484848485e-06, "loss": 0.0252, "num_tokens": 4929678.0, "reward": 0.8667252063751221, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8917012214660645, "reward_meter_std": 0.3005649745464325, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.29866674542427063, "reward_total_composite_mean": 0.8667252063751221, "reward_total_composite_std": 0.29866674542427063, "reward_total_mean": 0.8667252063751221, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8917012214660645, "rewards/meter/std": 0.3005649745464325, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.8667252063751221, "rewards/total_composite/std": 0.29866674542427063, "sampling/importance_sampling_ratio/max": 1.945900797843933, "sampling/importance_sampling_ratio/mean": 1.0148777961730957, "sampling/importance_sampling_ratio/min": 0.19834595918655396, "sampling/sampling_logp_difference/max": 1.6177425384521484, "sampling/sampling_logp_difference/mean": 0.05617932230234146, "step": 2184 }, { "clip_ratio/high_max": 0.0015625000232830644, "clip_ratio/high_mean": 0.0015625000232830644, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0015625000232830644, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.02709749317727983, "epoch": 0.08776157770012451, "frac_reward_zero_std": 0.0, "grad_norm": 3.390443801879883, "learning_rate": 3.3818181818181823e-06, "loss": 0.001, "num_tokens": 4931678.0, "reward": 0.6960331201553345, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7347736954689026, "reward_meter_std": 0.02520820125937462, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.04728476703166962, "reward_total_composite_mean": 0.6960331201553345, "reward_total_composite_std": 0.047284774482250214, "reward_total_mean": 0.6960331201553345, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7347736954689026, "rewards/meter/std": 0.02520820125937462, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.6960331201553345, "rewards/total_composite/std": 0.047284774482250214, "sampling/importance_sampling_ratio/max": 1.1883066892623901, "sampling/importance_sampling_ratio/mean": 1.000468134880066, "sampling/importance_sampling_ratio/min": 0.41800257563591003, "sampling/sampling_logp_difference/max": 0.8722677230834961, "sampling/sampling_logp_difference/mean": 0.0048207431100308895, "step": 2185 }, { "clip_ratio/high_max": 0.036382337333634496, "clip_ratio/high_mean": 0.036382337333634496, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/region_mean": 0.03816805162932724, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 75.125, "completions/mean_terminated_length": 75.125, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.3504325784742832, "epoch": 0.08780174318190946, "frac_reward_zero_std": 0.0, "grad_norm": 4.1797590255737305, "learning_rate": 3.378787878787879e-06, "loss": -0.0159, "num_tokens": 4933511.0, "reward": 0.9933065176010132, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9933065176010132, "reward_meter_std": 0.01265679020434618, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.012656783685088158, "reward_total_composite_mean": 0.9933065176010132, "reward_total_composite_std": 0.01265679020434618, "reward_total_mean": 0.9933065176010132, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9933065176010132, "rewards/meter/std": 0.01265679020434618, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9933065176010132, "rewards/total_composite/std": 0.01265679020434618, "sampling/importance_sampling_ratio/max": 1.9696723222732544, "sampling/importance_sampling_ratio/mean": 1.0102565288543701, "sampling/importance_sampling_ratio/min": 0.32066673040390015, "sampling/sampling_logp_difference/max": 1.1373529434204102, "sampling/sampling_logp_difference/mean": 0.03881411254405975, "step": 2186 }, { "clip_ratio/high_max": 0.02572961524128914, "clip_ratio/high_mean": 0.02572961524128914, "clip_ratio/low_mean": 0.015058388700708747, "clip_ratio/low_min": 0.015058388700708747, "clip_ratio/region_mean": 0.040788003941997886, "completions/clipped_ratio": 0.0, "completions/max_length": 354.0, "completions/max_terminated_length": 354.0, "completions/mean_length": 314.75, "completions/mean_terminated_length": 314.75, "completions/min_length": 297.0, "completions/min_terminated_length": 297.0, "entropy": 0.45917268842458725, "epoch": 0.08784190866369442, "frac_reward_zero_std": 0.0, "grad_norm": 1.9298547506332397, "learning_rate": 3.375757575757576e-06, "loss": -0.0221, "num_tokens": 4938469.0, "reward": 0.5502004623413086, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5803571939468384, "reward_count_adherence_std": 0.02525380253791809, "reward_meter_mean": 0.997092604637146, "reward_meter_std": 0.0013639559037983418, "reward_repeat_penalty_mean": 0.9509804248809814, "reward_repeat_penalty_std": 0.04682480916380882, "reward_std": 0.03388208895921707, "reward_total_composite_mean": 0.5502004623413086, "reward_total_composite_std": 0.03388208523392677, "reward_total_mean": 0.5502004623413086, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5803571939468384, "rewards/count_adherence/std": 0.02525380253791809, "rewards/meter/mean": 0.997092604637146, "rewards/meter/std": 0.0013639559037983418, "rewards/repeat_penalty/mean": 0.9509804248809814, "rewards/repeat_penalty/std": 0.04682480916380882, "rewards/total_composite/mean": 0.5502004623413086, "rewards/total_composite/std": 0.03388208523392677, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0096619129180908, "sampling/importance_sampling_ratio/min": 0.2924036681652069, "sampling/sampling_logp_difference/max": 1.2296199798583984, "sampling/sampling_logp_difference/mean": 0.04348405450582504, "step": 2187 }, { "clip_ratio/high_max": 0.006579951150342822, "clip_ratio/high_mean": 0.006579951150342822, "clip_ratio/low_mean": 0.0010775862028822303, "clip_ratio/low_min": 0.0010775862028822303, "clip_ratio/region_mean": 0.007657537353225052, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 115.0, "completions/mean_terminated_length": 115.0, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.07655041664838791, "epoch": 0.08788207414547937, "frac_reward_zero_std": 0.0, "grad_norm": 1.814806342124939, "learning_rate": 3.3727272727272732e-06, "loss": 0.0088, "num_tokens": 4940773.0, "reward": 0.9070344567298889, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959709644317627, "reward_meter_std": 0.00034307397436350584, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07348321378231049, "reward_total_composite_mean": 0.9070344567298889, "reward_total_composite_std": 0.07348322868347168, "reward_total_mean": 0.9070344567298889, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959709644317627, "rewards/meter/std": 0.00034307397436350584, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9070344567298889, "rewards/total_composite/std": 0.07348322868347168, "sampling/importance_sampling_ratio/max": 1.2936325073242188, "sampling/importance_sampling_ratio/mean": 0.9997696280479431, "sampling/importance_sampling_ratio/min": 0.27420806884765625, "sampling/sampling_logp_difference/max": 1.293868064880371, "sampling/sampling_logp_difference/mean": 0.015008204616606236, "step": 2188 }, { "clip_ratio/high_max": 0.0022321429569274187, "clip_ratio/high_mean": 0.0022321429569274187, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0022321429569274187, "completions/clipped_ratio": 0.0, "completions/max_length": 56.0, "completions/max_terminated_length": 56.0, "completions/mean_length": 56.0, "completions/mean_terminated_length": 56.0, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.05346089042723179, "epoch": 0.08792223962726432, "frac_reward_zero_std": 0.0, "grad_norm": 0.7317746877670288, "learning_rate": 3.36969696969697e-06, "loss": 0.0002, "num_tokens": 4942493.0, "reward": 0.9949744939804077, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949744939804077, "reward_meter_std": 3.23687017953489e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.236968768760562e-05, "reward_total_composite_mean": 0.9949744939804077, "reward_total_composite_std": 3.23687017953489e-05, "reward_total_mean": 0.9949744939804077, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949744939804077, "rewards/meter/std": 3.23687017953489e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949744939804077, "rewards/total_composite/std": 3.23687017953489e-05, "sampling/importance_sampling_ratio/max": 1.3790956735610962, "sampling/importance_sampling_ratio/mean": 1.003525972366333, "sampling/importance_sampling_ratio/min": 0.8439798355102539, "sampling/sampling_logp_difference/max": 0.32142794132232666, "sampling/sampling_logp_difference/mean": 0.00435090996325016, "step": 2189 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.011140820104628801, "clip_ratio/low_min": 0.011140820104628801, "clip_ratio/region_mean": 0.014928699005395174, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.375, "completions/mean_terminated_length": 33.375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.14891941286623478, "epoch": 0.08796240510904928, "frac_reward_zero_std": 0.0, "grad_norm": 3.3835291862487793, "learning_rate": 3.366666666666667e-06, "loss": 0.0157, "num_tokens": 4943784.0, "reward": 0.9737765789031982, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9737765789031982, "reward_meter_std": 0.008460394106805325, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008460381999611855, "reward_total_composite_mean": 0.9737765789031982, "reward_total_composite_std": 0.008460394106805325, "reward_total_mean": 0.9737765789031982, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9737765789031982, "rewards/meter/std": 0.008460394106805325, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9737765789031982, "rewards/total_composite/std": 0.008460394106805325, "sampling/importance_sampling_ratio/max": 1.51780366897583, "sampling/importance_sampling_ratio/mean": 1.0068951845169067, "sampling/importance_sampling_ratio/min": 0.33197662234306335, "sampling/sampling_logp_difference/max": 1.1026906967163086, "sampling/sampling_logp_difference/mean": 0.019433647394180298, "step": 2190 }, { "clip_ratio/high_max": 0.006637688144110143, "clip_ratio/high_mean": 0.006637688144110143, "clip_ratio/low_mean": 0.0021739129442721605, "clip_ratio/low_min": 0.0021739129442721605, "clip_ratio/region_mean": 0.008811601088382304, "completions/clipped_ratio": 0.0, "completions/max_length": 116.0, "completions/max_terminated_length": 116.0, "completions/mean_length": 113.75, "completions/mean_terminated_length": 113.75, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.09450818877667189, "epoch": 0.08800257059083423, "frac_reward_zero_std": 0.0, "grad_norm": 1.943103551864624, "learning_rate": 3.3636363636363637e-06, "loss": 0.0003, "num_tokens": 4946182.0, "reward": 0.9601929187774658, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.995752215385437, "reward_meter_std": 0.00029232812812551856, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06590110063552856, "reward_total_composite_mean": 0.9601929187774658, "reward_total_composite_std": 0.06590110063552856, "reward_total_mean": 0.9601929187774658, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.995752215385437, "rewards/meter/std": 0.00029232812812551856, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9601929187774658, "rewards/total_composite/std": 0.06590110063552856, "sampling/importance_sampling_ratio/max": 1.6537041664123535, "sampling/importance_sampling_ratio/mean": 1.0013221502304077, "sampling/importance_sampling_ratio/min": 0.2067027986049652, "sampling/sampling_logp_difference/max": 1.5764732360839844, "sampling/sampling_logp_difference/mean": 0.015357456170022488, "step": 2191 }, { "clip_ratio/high_max": 0.018826007028110325, "clip_ratio/high_mean": 0.018826007028110325, "clip_ratio/low_mean": 0.0012376237427815795, "clip_ratio/low_min": 0.0012376237427815795, "clip_ratio/region_mean": 0.020063630770891905, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 100.125, "completions/mean_terminated_length": 100.125, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.08774240221828222, "epoch": 0.08804273607261918, "frac_reward_zero_std": 0.0, "grad_norm": 3.4850220680236816, "learning_rate": 3.360606060606061e-06, "loss": 0.0058, "num_tokens": 4948311.0, "reward": 0.99900221824646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99900221824646, "reward_meter_std": 0.00022944888041820377, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00022944044030737132, "reward_total_composite_mean": 0.99900221824646, "reward_total_composite_std": 0.00022944888041820377, "reward_total_mean": 0.99900221824646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99900221824646, "rewards/meter/std": 0.00022944888041820377, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99900221824646, "rewards/total_composite/std": 0.00022944888041820377, "sampling/importance_sampling_ratio/max": 1.8097566366195679, "sampling/importance_sampling_ratio/mean": 0.9979958534240723, "sampling/importance_sampling_ratio/min": 0.27632972598075867, "sampling/sampling_logp_difference/max": 1.2861604690551758, "sampling/sampling_logp_difference/mean": 0.01750640757381916, "step": 2192 }, { "clip_ratio/high_max": 0.014968461007811129, "clip_ratio/high_mean": 0.014968461007811129, "clip_ratio/low_mean": 0.031126924557611346, "clip_ratio/low_min": 0.031126924557611346, "clip_ratio/region_mean": 0.046095385565422475, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 75.875, "completions/mean_terminated_length": 75.875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.5629463791847229, "epoch": 0.08808290155440414, "frac_reward_zero_std": 0.0, "grad_norm": 5.769536972045898, "learning_rate": 3.357575757575758e-06, "loss": 0.0298, "num_tokens": 4950310.0, "reward": 0.9790392518043518, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9790392518043518, "reward_meter_std": 0.02525605633854866, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.025256047025322914, "reward_total_composite_mean": 0.9790392518043518, "reward_total_composite_std": 0.02525605633854866, "reward_total_mean": 0.9790392518043518, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9790392518043518, "rewards/meter/std": 0.02525605633854866, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9790392518043518, "rewards/total_composite/std": 0.02525605633854866, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0126549005508423, "sampling/importance_sampling_ratio/min": 0.35265716910362244, "sampling/sampling_logp_difference/max": 1.042258858680725, "sampling/sampling_logp_difference/mean": 0.06429249048233032, "step": 2193 }, { "clip_ratio/high_max": 0.003597308532334864, "clip_ratio/high_mean": 0.003597308532334864, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.003597308532334864, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.25, "completions/mean_terminated_length": 69.25, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.04704178823158145, "epoch": 0.08812306703618909, "frac_reward_zero_std": 0.0, "grad_norm": 4.066121578216553, "learning_rate": 3.3545454545454547e-06, "loss": 0.0016, "num_tokens": 4952168.0, "reward": 0.9961166977882385, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961166977882385, "reward_meter_std": 0.0030905893072485924, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0030905804596841335, "reward_total_composite_mean": 0.9961166977882385, "reward_total_composite_std": 0.0030905893072485924, "reward_total_mean": 0.9961166977882385, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961166977882385, "rewards/meter/std": 0.0030905893072485924, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961166977882385, "rewards/total_composite/std": 0.0030905893072485924, "sampling/importance_sampling_ratio/max": 1.0906788110733032, "sampling/importance_sampling_ratio/mean": 0.9982797503471375, "sampling/importance_sampling_ratio/min": 0.49961915612220764, "sampling/sampling_logp_difference/max": 0.6939091682434082, "sampling/sampling_logp_difference/mean": 0.008842971175909042, "step": 2194 }, { "clip_ratio/high_max": 0.013268729439005256, "clip_ratio/high_mean": 0.013268729439005256, "clip_ratio/low_mean": 0.00645240826997906, "clip_ratio/low_min": 0.00645240826997906, "clip_ratio/region_mean": 0.019721137708984315, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 75.875, "completions/mean_terminated_length": 75.875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.3429943472146988, "epoch": 0.08816323251797405, "frac_reward_zero_std": 0.0, "grad_norm": 3.4402453899383545, "learning_rate": 3.351515151515152e-06, "loss": 0.0008, "num_tokens": 4953959.0, "reward": 0.998624861240387, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998624861240387, "reward_meter_std": 0.0006542301853187382, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006542490446008742, "reward_total_composite_mean": 0.998624861240387, "reward_total_composite_std": 0.0006542301853187382, "reward_total_mean": 0.998624861240387, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998624861240387, "rewards/meter/std": 0.0006542301853187382, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998624861240387, "rewards/total_composite/std": 0.0006542301853187382, "sampling/importance_sampling_ratio/max": 1.4867199659347534, "sampling/importance_sampling_ratio/mean": 1.0039963722229004, "sampling/importance_sampling_ratio/min": 0.3405395448207855, "sampling/sampling_logp_difference/max": 1.0772240161895752, "sampling/sampling_logp_difference/mean": 0.04149501770734787, "step": 2195 }, { "clip_ratio/high_max": 0.003846153849735856, "clip_ratio/high_mean": 0.003846153849735856, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.003846153849735856, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.625, "completions/mean_terminated_length": 65.625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.04586625238880515, "epoch": 0.088203397999759, "frac_reward_zero_std": 0.0, "grad_norm": 1.1731183528900146, "learning_rate": 3.3484848484848487e-06, "loss": 0.0022, "num_tokens": 4955804.0, "reward": 0.9991217255592346, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991217255592346, "reward_meter_std": 3.6965378967579454e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.6982291931053624e-05, "reward_total_composite_mean": 0.9991217255592346, "reward_total_composite_std": 3.6965378967579454e-05, "reward_total_mean": 0.9991217255592346, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991217255592346, "rewards/meter/std": 3.6965378967579454e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991217255592346, "rewards/total_composite/std": 3.6965378967579454e-05, "sampling/importance_sampling_ratio/max": 1.6666737794876099, "sampling/importance_sampling_ratio/mean": 1.0017669200897217, "sampling/importance_sampling_ratio/min": 0.7030104994773865, "sampling/sampling_logp_difference/max": 0.5108299255371094, "sampling/sampling_logp_difference/mean": 0.007183389738202095, "step": 2196 }, { "clip_ratio/high_max": 0.011807307717390358, "clip_ratio/high_mean": 0.011807307717390358, "clip_ratio/low_mean": 0.010880606481805444, "clip_ratio/low_min": 0.010880606481805444, "clip_ratio/region_mean": 0.022687914199195802, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 75.5, "completions/mean_terminated_length": 75.5, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.20823358744382858, "epoch": 0.08824356348154395, "frac_reward_zero_std": 0.0, "grad_norm": 9.319520950317383, "learning_rate": 3.3454545454545456e-06, "loss": 0.0579, "num_tokens": 4957640.0, "reward": 0.9673671126365662, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9673671126365662, "reward_meter_std": 0.03739238530397415, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03739236667752266, "reward_total_composite_mean": 0.9673671126365662, "reward_total_composite_std": 0.03739238530397415, "reward_total_mean": 0.9673671126365662, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9673671126365662, "rewards/meter/std": 0.03739238530397415, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9673671126365662, "rewards/total_composite/std": 0.03739238530397415, "sampling/importance_sampling_ratio/max": 1.8825844526290894, "sampling/importance_sampling_ratio/mean": 1.0026880502700806, "sampling/importance_sampling_ratio/min": 0.17287828028202057, "sampling/sampling_logp_difference/max": 1.7551674842834473, "sampling/sampling_logp_difference/mean": 0.032729994505643845, "step": 2197 }, { "clip_ratio/high_max": 0.006831709994003177, "clip_ratio/high_mean": 0.006831709994003177, "clip_ratio/low_mean": 0.0032051282469183207, "clip_ratio/low_min": 0.0032051282469183207, "clip_ratio/region_mean": 0.010036838240921497, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 75.25, "completions/mean_terminated_length": 75.25, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.24795558489859104, "epoch": 0.08828372896332891, "frac_reward_zero_std": 0.0, "grad_norm": 3.934696912765503, "learning_rate": 3.3424242424242424e-06, "loss": 0.0073, "num_tokens": 4959602.0, "reward": 0.9990965127944946, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990965127944946, "reward_meter_std": 0.00029607766191475093, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00029608141630887985, "reward_total_composite_mean": 0.9990965127944946, "reward_total_composite_std": 0.00029607766191475093, "reward_total_mean": 0.9990965127944946, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990965127944946, "rewards/meter/std": 0.00029607766191475093, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990965127944946, "rewards/total_composite/std": 0.00029607766191475093, "sampling/importance_sampling_ratio/max": 1.3668806552886963, "sampling/importance_sampling_ratio/mean": 1.0073840618133545, "sampling/importance_sampling_ratio/min": 0.4389890432357788, "sampling/sampling_logp_difference/max": 0.8232808113098145, "sampling/sampling_logp_difference/mean": 0.021792318671941757, "step": 2198 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/region_mean": 0.0036231884732842445, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.025351210730150342, "epoch": 0.08832389444511386, "frac_reward_zero_std": 0.0, "grad_norm": 0.02073613740503788, "learning_rate": 3.3393939393939397e-06, "loss": 0.0001, "num_tokens": 4961586.0, "reward": 0.9973071813583374, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973071813583374, "reward_meter_std": 2.9068671665299917e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.895276793424273e-06, "reward_total_composite_mean": 0.9973071813583374, "reward_total_composite_std": 2.9068671665299917e-06, "reward_total_mean": 0.9973071813583374, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973071813583374, "rewards/meter/std": 2.9068671665299917e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973071813583374, "rewards/total_composite/std": 2.9068671665299917e-06, "sampling/importance_sampling_ratio/max": 1.067936897277832, "sampling/importance_sampling_ratio/mean": 1.0010336637496948, "sampling/importance_sampling_ratio/min": 0.8587086200714111, "sampling/sampling_logp_difference/max": 0.15232563018798828, "sampling/sampling_logp_difference/mean": 0.0027781545650213957, "step": 2199 }, { "clip_ratio/high_max": 0.03107945283409208, "clip_ratio/high_mean": 0.03107945283409208, "clip_ratio/low_mean": 0.0031645570416003466, "clip_ratio/low_min": 0.0031645570416003466, "clip_ratio/region_mean": 0.03424400987569243, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 80.125, "completions/mean_terminated_length": 80.125, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.3972342722117901, "epoch": 0.08836405992689882, "frac_reward_zero_std": 0.0, "grad_norm": 3.2771596908569336, "learning_rate": 3.3363636363636365e-06, "loss": -0.0049, "num_tokens": 4963619.0, "reward": 0.8722573518753052, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969666600227356, "reward_meter_std": 0.001407405361533165, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35244786739349365, "reward_total_composite_mean": 0.8722573518753052, "reward_total_composite_std": 0.35244786739349365, "reward_total_mean": 0.8722573518753052, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969666600227356, "rewards/meter/std": 0.001407405361533165, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8722573518753052, "rewards/total_composite/std": 0.35244786739349365, "sampling/importance_sampling_ratio/max": 1.6984498500823975, "sampling/importance_sampling_ratio/mean": 1.0073878765106201, "sampling/importance_sampling_ratio/min": 0.2997937798500061, "sampling/sampling_logp_difference/max": 1.204660415649414, "sampling/sampling_logp_difference/mean": 0.038304030895233154, "step": 2200 }, { "epoch": 0.08836405992689882, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 339.15384615384613, "eval_completions/max_terminated_length": 339.15384615384613, "eval_completions/mean_length": 187.77884615384616, "eval_completions/mean_terminated_length": 187.77884615384616, "eval_completions/min_length": 62.69230769230769, "eval_completions/min_terminated_length": 62.69230769230769, "eval_entropy": 0.1745165208211312, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 4963619.0, "eval_reward": 0.5540328117517325, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.8735798276387728, "eval_reward_count_adherence_std": 0.14306257034723574, "eval_reward_meter_mean": 0.722443516437824, "eval_reward_meter_std": 0.41069405812483567, "eval_reward_repeat_penalty_mean": 0.8453576381389911, "eval_reward_repeat_penalty_std": 0.13899515225337103, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5540328117517325, "eval_reward_total_composite_std": 0.3618074002174231, "eval_reward_total_mean": 0.5540328117517325, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.8735798276387728, "eval_rewards/count_adherence/std": 0.14306257034723574, "eval_rewards/meter/mean": 0.722443516437824, "eval_rewards/meter/std": 0.41069405812483567, "eval_rewards/repeat_penalty/mean": 0.8453576381389911, "eval_rewards/repeat_penalty/std": 0.13899515225337103, "eval_rewards/total_composite/mean": 0.5540328117517325, "eval_rewards/total_composite/std": 0.3618074002174231, "eval_runtime": 65.1539, "eval_samples_per_second": 1.596, "eval_sampling/importance_sampling_ratio/max": 1.4119950441213756, "eval_sampling/importance_sampling_ratio/mean": 1.0049138986147368, "eval_sampling/importance_sampling_ratio/min": 0.3679314060853078, "eval_sampling/sampling_logp_difference/max": 1.04065675001878, "eval_sampling/sampling_logp_difference/mean": 0.01593467600357074, "eval_steps_per_second": 0.2, "step": 2200 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.029302328824996948, "epoch": 0.08840422540868377, "frac_reward_zero_std": 0.0, "grad_norm": 2.442410707473755, "learning_rate": 3.3333333333333333e-06, "loss": 0.0037, "num_tokens": 4965596.0, "reward": 0.9970394968986511, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970394968986511, "reward_meter_std": 0.0007004055660218, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007004134822636843, "reward_total_composite_mean": 0.9970394968986511, "reward_total_composite_std": 0.0007004055660218, "reward_total_mean": 0.9970394968986511, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970394968986511, "rewards/meter/std": 0.0007004055660218, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970394968986511, "rewards/total_composite/std": 0.0007004055660218, "sampling/importance_sampling_ratio/max": 1.3967480659484863, "sampling/importance_sampling_ratio/mean": 1.001259207725525, "sampling/importance_sampling_ratio/min": 0.48868146538734436, "sampling/sampling_logp_difference/max": 0.7160444259643555, "sampling/sampling_logp_difference/mean": 0.005273114424198866, "step": 2201 }, { "clip_ratio/high_max": 0.020743532106280327, "clip_ratio/high_mean": 0.020743532106280327, "clip_ratio/low_mean": 0.014749160269275308, "clip_ratio/low_min": 0.014749160269275308, "clip_ratio/region_mean": 0.035492692375555634, "completions/clipped_ratio": 0.0, "completions/max_length": 285.0, "completions/max_terminated_length": 285.0, "completions/mean_length": 277.875, "completions/mean_terminated_length": 277.875, "completions/min_length": 273.0, "completions/min_terminated_length": 273.0, "entropy": 0.4109150692820549, "epoch": 0.08844439089046872, "frac_reward_zero_std": 0.0, "grad_norm": 2.4785351753234863, "learning_rate": 3.3303030303030306e-06, "loss": 0.0069, "num_tokens": 4969547.0, "reward": 0.8308628797531128, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997477114200592, "reward_meter_std": 0.001005422556772828, "reward_repeat_penalty_mean": 0.9519230723381042, "reward_repeat_penalty_std": 0.05723259970545769, "reward_std": 0.05055440962314606, "reward_total_composite_mean": 0.8308628797531128, "reward_total_composite_std": 0.05055442824959755, "reward_total_mean": 0.8308628797531128, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997477114200592, "rewards/meter/std": 0.001005422556772828, "rewards/repeat_penalty/mean": 0.9519230723381042, "rewards/repeat_penalty/std": 0.05723259970545769, "rewards/total_composite/mean": 0.8308628797531128, "rewards/total_composite/std": 0.05055442824959755, "sampling/importance_sampling_ratio/max": 1.8519618511199951, "sampling/importance_sampling_ratio/mean": 1.0075916051864624, "sampling/importance_sampling_ratio/min": 0.25140586495399475, "sampling/sampling_logp_difference/max": 1.3806867599487305, "sampling/sampling_logp_difference/mean": 0.046521060168743134, "step": 2202 }, { "clip_ratio/high_max": 0.030337184784002602, "clip_ratio/high_mean": 0.030337184784002602, "clip_ratio/low_mean": 0.0050675676902756095, "clip_ratio/low_min": 0.0050675676902756095, "clip_ratio/region_mean": 0.03540475247427821, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 73.875, "completions/mean_terminated_length": 73.875, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.46222241409122944, "epoch": 0.08848455637225369, "frac_reward_zero_std": 0.0, "grad_norm": 3.297503709793091, "learning_rate": 3.3272727272727274e-06, "loss": 0.0081, "num_tokens": 4971474.0, "reward": 0.9979462623596191, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979462623596191, "reward_meter_std": 0.002537182066589594, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002537175314500928, "reward_total_composite_mean": 0.9979462623596191, "reward_total_composite_std": 0.002537182066589594, "reward_total_mean": 0.9979462623596191, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979462623596191, "rewards/meter/std": 0.002537182066589594, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979462623596191, "rewards/total_composite/std": 0.002537182066589594, "sampling/importance_sampling_ratio/max": 1.8745375871658325, "sampling/importance_sampling_ratio/mean": 1.004874587059021, "sampling/importance_sampling_ratio/min": 0.17576563358306885, "sampling/sampling_logp_difference/max": 1.7386038303375244, "sampling/sampling_logp_difference/mean": 0.060300085693597794, "step": 2203 }, { "clip_ratio/high_max": 0.014851224608719349, "clip_ratio/high_mean": 0.014851224608719349, "clip_ratio/low_mean": 0.014909796707797796, "clip_ratio/low_min": 0.014909796707797796, "clip_ratio/region_mean": 0.029761021316517144, "completions/clipped_ratio": 0.0, "completions/max_length": 449.0, "completions/max_terminated_length": 449.0, "completions/mean_length": 410.875, "completions/mean_terminated_length": 410.875, "completions/min_length": 380.0, "completions/min_terminated_length": 380.0, "entropy": 0.3659870997071266, "epoch": 0.08852472185403865, "frac_reward_zero_std": 0.0, "grad_norm": 1.867836833000183, "learning_rate": 3.3242424242424242e-06, "loss": -0.0316, "num_tokens": 4976337.0, "reward": 0.4941902458667755, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6029411554336548, "reward_count_adherence_std": 0.027230001986026764, "reward_meter_mean": 0.9976021647453308, "reward_meter_std": 0.0008352479781024158, "reward_repeat_penalty_mean": 0.8183584213256836, "reward_repeat_penalty_std": 0.10606583207845688, "reward_std": 0.08418715000152588, "reward_total_composite_mean": 0.4941902458667755, "reward_total_composite_std": 0.08418715745210648, "reward_total_mean": 0.4941902458667755, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6029411554336548, "rewards/count_adherence/std": 0.027230001986026764, "rewards/meter/mean": 0.9976021647453308, "rewards/meter/std": 0.0008352479781024158, "rewards/repeat_penalty/mean": 0.8183584213256836, "rewards/repeat_penalty/std": 0.10606583207845688, "rewards/total_composite/mean": 0.4941902458667755, "rewards/total_composite/std": 0.08418715745210648, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0103265047073364, "sampling/importance_sampling_ratio/min": 0.0014744381187483668, "sampling/sampling_logp_difference/max": 6.5194783210754395, "sampling/sampling_logp_difference/mean": 0.0467507466673851, "step": 2204 }, { "clip_ratio/high_max": 0.0022346453624777496, "clip_ratio/high_mean": 0.0022346453624777496, "clip_ratio/low_mean": 0.0005555555690079927, "clip_ratio/low_min": 0.0005555555690079927, "clip_ratio/region_mean": 0.0027902009314857423, "completions/clipped_ratio": 0.0, "completions/max_length": 225.0, "completions/max_terminated_length": 225.0, "completions/mean_length": 223.625, "completions/mean_terminated_length": 223.625, "completions/min_length": 223.0, "completions/min_terminated_length": 223.0, "entropy": 0.02647018409334123, "epoch": 0.0885648873358236, "frac_reward_zero_std": 0.0, "grad_norm": 1.170793056488037, "learning_rate": 3.321212121212121e-06, "loss": 0.0016, "num_tokens": 4979646.0, "reward": 0.5902398228645325, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9960297346115112, "reward_meter_std": 0.0002207064680987969, "reward_repeat_penalty_mean": 0.6666666865348816, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00013079482596367598, "reward_total_composite_mean": 0.5902398228645325, "reward_total_composite_std": 0.00013078592019155622, "reward_total_mean": 0.5902398228645325, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9960297346115112, "rewards/meter/std": 0.0002207064680987969, "rewards/repeat_penalty/mean": 0.6666666865348816, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5902398228645325, "rewards/total_composite/std": 0.00013078592019155622, "sampling/importance_sampling_ratio/max": 1.5353397130966187, "sampling/importance_sampling_ratio/mean": 1.0011334419250488, "sampling/importance_sampling_ratio/min": 0.27995866537094116, "sampling/sampling_logp_difference/max": 1.2731132507324219, "sampling/sampling_logp_difference/mean": 0.005620577838271856, "step": 2205 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.009517844882793725, "epoch": 0.08860505281760855, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.3181818181818188e-06, "loss": 0.0, "num_tokens": 4981398.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0276442766189575, "sampling/importance_sampling_ratio/mean": 1.0007154941558838, "sampling/importance_sampling_ratio/min": 0.9798296689987183, "sampling/sampling_logp_difference/max": 0.027269084006547928, "sampling/sampling_logp_difference/mean": 0.0008829182479530573, "step": 2206 }, { "clip_ratio/high_max": 0.019536115461960435, "clip_ratio/high_mean": 0.019536115461960435, "clip_ratio/low_mean": 0.0101922721369192, "clip_ratio/low_min": 0.0101922721369192, "clip_ratio/region_mean": 0.029728387598879635, "completions/clipped_ratio": 0.0, "completions/max_length": 396.0, "completions/max_terminated_length": 396.0, "completions/mean_length": 377.5, "completions/mean_terminated_length": 377.5, "completions/min_length": 362.0, "completions/min_terminated_length": 362.0, "entropy": 0.3299342021346092, "epoch": 0.08864521829939351, "frac_reward_zero_std": 0.0, "grad_norm": 1.6692476272583008, "learning_rate": 3.3151515151515156e-06, "loss": -0.0214, "num_tokens": 4986186.0, "reward": 0.5530020594596863, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.625, "reward_count_adherence_std": 0.0345032773911953, "reward_meter_mean": 0.9973218441009521, "reward_meter_std": 0.000687557680066675, "reward_repeat_penalty_mean": 0.8846749067306519, "reward_repeat_penalty_std": 0.11391878128051758, "reward_std": 0.08831970393657684, "reward_total_composite_mean": 0.5530020594596863, "reward_total_composite_std": 0.08831970393657684, "reward_total_mean": 0.5530020594596863, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.625, "rewards/count_adherence/std": 0.0345032773911953, "rewards/meter/mean": 0.9973218441009521, "rewards/meter/std": 0.000687557680066675, "rewards/repeat_penalty/mean": 0.8846749067306519, "rewards/repeat_penalty/std": 0.11391878128051758, "rewards/total_composite/mean": 0.5530020594596863, "rewards/total_composite/std": 0.08831970393657684, "sampling/importance_sampling_ratio/max": 1.8353766202926636, "sampling/importance_sampling_ratio/mean": 1.0060136318206787, "sampling/importance_sampling_ratio/min": 0.10464474558830261, "sampling/sampling_logp_difference/max": 2.2571840286254883, "sampling/sampling_logp_difference/mean": 0.03707325831055641, "step": 2207 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.00906775618204847, "epoch": 0.08868538378117846, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.3121212121212124e-06, "loss": 0.0, "num_tokens": 4988010.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.028230905532837, "sampling/importance_sampling_ratio/mean": 1.0008260011672974, "sampling/importance_sampling_ratio/min": 0.9878785610198975, "sampling/sampling_logp_difference/max": 0.027839738875627518, "sampling/sampling_logp_difference/mean": 0.0008998184930533171, "step": 2208 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.011399149021599442, "epoch": 0.08872554926296342, "frac_reward_zero_std": 0.0, "grad_norm": 2.7090210914611816, "learning_rate": 3.3090909090909097e-06, "loss": -0.0012, "num_tokens": 4989794.0, "reward": 0.7749799489974976, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7749799489974976, "reward_meter_std": 0.033681225031614304, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.033681221306324005, "reward_total_composite_mean": 0.7749799489974976, "reward_total_composite_std": 0.033681225031614304, "reward_total_mean": 0.7749799489974976, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7749799489974976, "rewards/meter/std": 0.033681225031614304, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7749799489974976, "rewards/total_composite/std": 0.033681225031614304, "sampling/importance_sampling_ratio/max": 1.0404185056686401, "sampling/importance_sampling_ratio/mean": 0.9990789890289307, "sampling/importance_sampling_ratio/min": 0.4408467411994934, "sampling/sampling_logp_difference/max": 0.8190579414367676, "sampling/sampling_logp_difference/mean": 0.00368741643615067, "step": 2209 }, { "clip_ratio/high_max": 0.01074601011350751, "clip_ratio/high_mean": 0.01074601011350751, "clip_ratio/low_mean": 0.018732663709670305, "clip_ratio/low_min": 0.018732663709670305, "clip_ratio/region_mean": 0.029478673823177814, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 80.375, "completions/mean_terminated_length": 80.375, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.3479005917906761, "epoch": 0.08876571474474837, "frac_reward_zero_std": 0.0, "grad_norm": 6.713454723358154, "learning_rate": 3.3060606060606065e-06, "loss": 0.0121, "num_tokens": 4991661.0, "reward": 0.9961856007575989, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961856007575989, "reward_meter_std": 0.0025163544341921806, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0025163493119180202, "reward_total_composite_mean": 0.9961856007575989, "reward_total_composite_std": 0.0025163544341921806, "reward_total_mean": 0.9961856007575989, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961856007575989, "rewards/meter/std": 0.0025163544341921806, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961856007575989, "rewards/total_composite/std": 0.0025163544341921806, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0097302198410034, "sampling/importance_sampling_ratio/min": 0.2356281578540802, "sampling/sampling_logp_difference/max": 1.445500373840332, "sampling/sampling_logp_difference/mean": 0.039697032421827316, "step": 2210 }, { "clip_ratio/high_max": 0.0045045046135783195, "clip_ratio/high_mean": 0.0045045046135783195, "clip_ratio/low_mean": 0.0022522523067891598, "clip_ratio/low_min": 0.0022522523067891598, "clip_ratio/region_mean": 0.006756756920367479, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 111.0, "completions/mean_terminated_length": 111.0, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.03893151739612222, "epoch": 0.08880588022653332, "frac_reward_zero_std": 0.0, "grad_norm": 0.7238816022872925, "learning_rate": 3.3030303030303033e-06, "loss": 0.0013, "num_tokens": 4993997.0, "reward": 0.8533860445022583, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956170320510864, "reward_meter_std": 6.146137457108125e-05, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 5.2673178288387135e-05, "reward_total_composite_mean": 0.8533860445022583, "reward_total_composite_std": 5.2672676247311756e-05, "reward_total_mean": 0.8533860445022583, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956170320510864, "rewards/meter/std": 6.146137457108125e-05, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8533860445022583, "rewards/total_composite/std": 5.2672676247311756e-05, "sampling/importance_sampling_ratio/max": 1.6665387153625488, "sampling/importance_sampling_ratio/mean": 0.9998362064361572, "sampling/importance_sampling_ratio/min": 0.4735487401485443, "sampling/sampling_logp_difference/max": 0.7475004196166992, "sampling/sampling_logp_difference/mean": 0.007354200817644596, "step": 2211 }, { "clip_ratio/high_max": 0.006823194562457502, "clip_ratio/high_mean": 0.006823194562457502, "clip_ratio/low_mean": 0.005421726265922189, "clip_ratio/low_min": 0.005421726265922189, "clip_ratio/region_mean": 0.01224492082837969, "completions/clipped_ratio": 0.0, "completions/max_length": 277.0, "completions/max_terminated_length": 277.0, "completions/mean_length": 276.0, "completions/mean_terminated_length": 276.0, "completions/min_length": 274.0, "completions/min_terminated_length": 274.0, "entropy": 0.06808455288410187, "epoch": 0.08884604570831828, "frac_reward_zero_std": 0.0, "grad_norm": 1.9566975831985474, "learning_rate": 3.3000000000000006e-06, "loss": 0.0064, "num_tokens": 4998173.0, "reward": 0.5762275457382202, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6666666865348816, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973146319389343, "reward_meter_std": 0.0006697024800814688, "reward_repeat_penalty_mean": 0.8666666746139526, "reward_repeat_penalty_std": 0.06172133609652519, "reward_std": 0.041072458028793335, "reward_total_composite_mean": 0.5762275457382202, "reward_total_composite_std": 0.04107243940234184, "reward_total_mean": 0.5762275457382202, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6666666865348816, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973146319389343, "rewards/meter/std": 0.0006697024800814688, "rewards/repeat_penalty/mean": 0.8666666746139526, "rewards/repeat_penalty/std": 0.06172133609652519, "rewards/total_composite/mean": 0.5762275457382202, "rewards/total_composite/std": 0.04107243940234184, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0007983446121216, "sampling/importance_sampling_ratio/min": 0.17606841027736664, "sampling/sampling_logp_difference/max": 1.8627681732177734, "sampling/sampling_logp_difference/mean": 0.015320011414587498, "step": 2212 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.004421527351951227, "epoch": 0.08888621119010323, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.2969696969696974e-06, "loss": 0.0, "num_tokens": 4999621.0, "reward": 0.9929623007774353, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9929623007774353, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9929623007774353, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9929623007774353, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9929623007774353, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929623007774353, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.023474931716919, "sampling/importance_sampling_ratio/mean": 1.000752329826355, "sampling/importance_sampling_ratio/min": 0.9999908804893494, "sampling/sampling_logp_difference/max": 0.023203659802675247, "sampling/sampling_logp_difference/mean": 0.000746682460885495, "step": 2213 }, { "clip_ratio/high_max": 0.015862172469496727, "clip_ratio/high_mean": 0.015862172469496727, "clip_ratio/low_mean": 0.025091978488489985, "clip_ratio/low_min": 0.025091978488489985, "clip_ratio/region_mean": 0.04095415095798671, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 306.0, "completions/mean_terminated_length": 306.0, "completions/min_length": 291.0, "completions/min_terminated_length": 291.0, "entropy": 0.3657448794692755, "epoch": 0.08892637667188819, "frac_reward_zero_std": 0.0, "grad_norm": 2.7332818508148193, "learning_rate": 3.2939393939393943e-06, "loss": 0.0124, "num_tokens": 5003989.0, "reward": 0.5923222303390503, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6477272510528564, "reward_count_adherence_std": 0.03214123100042343, "reward_meter_mean": 0.996366024017334, "reward_meter_std": 0.004282170441001654, "reward_repeat_penalty_mean": 0.9187728762626648, "reward_repeat_penalty_std": 0.04368516802787781, "reward_std": 0.027357440441846848, "reward_total_composite_mean": 0.5923222303390503, "reward_total_composite_std": 0.027357451617717743, "reward_total_mean": 0.5923222303390503, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6477272510528564, "rewards/count_adherence/std": 0.03214123100042343, "rewards/meter/mean": 0.996366024017334, "rewards/meter/std": 0.004282170441001654, "rewards/repeat_penalty/mean": 0.9187728762626648, "rewards/repeat_penalty/std": 0.04368516802787781, "rewards/total_composite/mean": 0.5923222303390503, "rewards/total_composite/std": 0.027357451617717743, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0054079294204712, "sampling/importance_sampling_ratio/min": 0.1374780386686325, "sampling/sampling_logp_difference/max": 1.9842910766601562, "sampling/sampling_logp_difference/mean": 0.04633094370365143, "step": 2214 }, { "clip_ratio/high_max": 0.030406045261770487, "clip_ratio/high_mean": 0.030406045261770487, "clip_ratio/low_mean": 0.011416577035561204, "clip_ratio/low_min": 0.011416577035561204, "clip_ratio/region_mean": 0.04182262229733169, "completions/clipped_ratio": 0.0, "completions/max_length": 127.0, "completions/max_terminated_length": 127.0, "completions/mean_length": 122.25, "completions/mean_terminated_length": 122.25, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.3795941434800625, "epoch": 0.08896654215367314, "frac_reward_zero_std": 0.0, "grad_norm": 3.505354404449463, "learning_rate": 3.290909090909091e-06, "loss": -0.0075, "num_tokens": 5006191.0, "reward": 0.9975504875183105, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975504875183105, "reward_meter_std": 0.0010059267515316606, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001005922444164753, "reward_total_composite_mean": 0.9975504875183105, "reward_total_composite_std": 0.0010059267515316606, "reward_total_mean": 0.9975504875183105, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975504875183105, "rewards/meter/std": 0.0010059267515316606, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975504875183105, "rewards/total_composite/std": 0.0010059267515316606, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016958713531494, "sampling/importance_sampling_ratio/min": 0.3264273405075073, "sampling/sampling_logp_difference/max": 1.1195478439331055, "sampling/sampling_logp_difference/mean": 0.04226119443774223, "step": 2215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.008104118285700679, "epoch": 0.0890067076354581, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.2878787878787883e-06, "loss": 0.0, "num_tokens": 5007999.0, "reward": 0.9992372989654541, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992372989654541, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992372989654541, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992372989654541, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992372989654541, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992372989654541, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0303282737731934, "sampling/importance_sampling_ratio/mean": 1.0005970001220703, "sampling/importance_sampling_ratio/min": 0.9758114814758301, "sampling/sampling_logp_difference/max": 0.029877394437789917, "sampling/sampling_logp_difference/mean": 0.0007629028987139463, "step": 2216 }, { "clip_ratio/high_max": 0.014177569886669517, "clip_ratio/high_mean": 0.014177569886669517, "clip_ratio/low_mean": 0.005999947316013277, "clip_ratio/low_min": 0.005999947316013277, "clip_ratio/region_mean": 0.020177517202682793, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 108.25, "completions/mean_terminated_length": 108.25, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.14850023202598095, "epoch": 0.08904687311724305, "frac_reward_zero_std": 0.0, "grad_norm": 3.7353017330169678, "learning_rate": 3.284848484848485e-06, "loss": -0.0229, "num_tokens": 5010225.0, "reward": 0.8596517443656921, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9519559144973755, "reward_meter_std": 0.09988751262426376, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.15026216208934784, "reward_total_composite_mean": 0.8596517443656921, "reward_total_composite_std": 0.15026217699050903, "reward_total_mean": 0.8596517443656921, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9519559144973755, "rewards/meter/std": 0.09988751262426376, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8596517443656921, "rewards/total_composite/std": 0.15026217699050903, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003779649734497, "sampling/importance_sampling_ratio/min": 0.36630338430404663, "sampling/sampling_logp_difference/max": 1.004293441772461, "sampling/sampling_logp_difference/mean": 0.02141641452908516, "step": 2217 }, { "clip_ratio/high_max": 0.017237887950614095, "clip_ratio/high_mean": 0.017237887950614095, "clip_ratio/low_mean": 0.011183922295458615, "clip_ratio/low_min": 0.011183922295458615, "clip_ratio/region_mean": 0.02842181024607271, "completions/clipped_ratio": 0.0, "completions/max_length": 487.0, "completions/max_terminated_length": 487.0, "completions/mean_length": 474.625, "completions/mean_terminated_length": 474.625, "completions/min_length": 460.0, "completions/min_terminated_length": 460.0, "entropy": 0.28974736854434013, "epoch": 0.089087038599028, "frac_reward_zero_std": 0.0, "grad_norm": 1.9956766366958618, "learning_rate": 3.281818181818182e-06, "loss": -0.003, "num_tokens": 5015622.0, "reward": 0.5244148969650269, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5986841917037964, "reward_count_adherence_std": 0.027239417657256126, "reward_meter_mean": 0.9962097406387329, "reward_meter_std": 0.003824215615168214, "reward_repeat_penalty_mean": 0.8791407942771912, "reward_repeat_penalty_std": 0.11821687966585159, "reward_std": 0.07367353141307831, "reward_total_composite_mean": 0.5244148969650269, "reward_total_composite_std": 0.0736735388636589, "reward_total_mean": 0.5244148969650269, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5986841917037964, "rewards/count_adherence/std": 0.027239417657256126, "rewards/meter/mean": 0.9962097406387329, "rewards/meter/std": 0.003824215615168214, "rewards/repeat_penalty/mean": 0.8791407942771912, "rewards/repeat_penalty/std": 0.11821687966585159, "rewards/total_composite/mean": 0.5244148969650269, "rewards/total_composite/std": 0.0736735388636589, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007934808731079, "sampling/importance_sampling_ratio/min": 0.15779319405555725, "sampling/sampling_logp_difference/max": 1.8464699983596802, "sampling/sampling_logp_difference/mean": 0.04105058312416077, "step": 2218 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.125, "completions/mean_terminated_length": 64.125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.01567571924533695, "epoch": 0.08912720408081296, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.2787878787878793e-06, "loss": 0.0, "num_tokens": 5017335.0, "reward": 0.9992372989654541, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992372989654541, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992372989654541, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992372989654541, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992372989654541, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992372989654541, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.1355178356170654, "sampling/importance_sampling_ratio/mean": 1.000244140625, "sampling/importance_sampling_ratio/min": 0.7368006110191345, "sampling/sampling_logp_difference/max": 0.3054380416870117, "sampling/sampling_logp_difference/mean": 0.0031146432738751173, "step": 2219 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 137.0, "completions/mean_terminated_length": 137.0, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.018264403683133423, "epoch": 0.08916736956259791, "frac_reward_zero_std": 0.0, "grad_norm": 2.9601826667785645, "learning_rate": 3.275757575757576e-06, "loss": 0.0018, "num_tokens": 5019823.0, "reward": 0.8546586632728577, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971017837524414, "reward_meter_std": 0.0003135971783194691, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00026877920026890934, "reward_total_composite_mean": 0.8546586632728577, "reward_total_composite_std": 0.0002687963715288788, "reward_total_mean": 0.8546586632728577, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971017837524414, "rewards/meter/std": 0.0003135971783194691, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8546586632728577, "rewards/total_composite/std": 0.0002687963715288788, "sampling/importance_sampling_ratio/max": 1.3408164978027344, "sampling/importance_sampling_ratio/mean": 0.9999095797538757, "sampling/importance_sampling_ratio/min": 0.565025806427002, "sampling/sampling_logp_difference/max": 0.5708838701248169, "sampling/sampling_logp_difference/mean": 0.003638867987319827, "step": 2220 }, { "clip_ratio/high_max": 0.0015060240402817726, "clip_ratio/high_mean": 0.0015060240402817726, "clip_ratio/low_mean": 0.0015243901871144772, "clip_ratio/low_min": 0.0015243901871144772, "clip_ratio/region_mean": 0.0030304142273962498, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 82.875, "completions/mean_terminated_length": 82.875, "completions/min_length": 82.0, "completions/min_terminated_length": 82.0, "entropy": 0.03914514696225524, "epoch": 0.08920753504438286, "frac_reward_zero_std": 0.0, "grad_norm": 0.8054800033569336, "learning_rate": 3.272727272727273e-06, "loss": -0.0001, "num_tokens": 5021798.0, "reward": 0.9953393936157227, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953393936157227, "reward_meter_std": 4.823669223696925e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.822280607186258e-05, "reward_total_composite_mean": 0.9953393936157227, "reward_total_composite_std": 4.823669223696925e-05, "reward_total_mean": 0.9953393936157227, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953393936157227, "rewards/meter/std": 4.823669223696925e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953393936157227, "rewards/total_composite/std": 4.823669223696925e-05, "sampling/importance_sampling_ratio/max": 1.2980178594589233, "sampling/importance_sampling_ratio/mean": 0.99955153465271, "sampling/importance_sampling_ratio/min": 0.4006454050540924, "sampling/sampling_logp_difference/max": 0.9146785736083984, "sampling/sampling_logp_difference/mean": 0.009308185428380966, "step": 2221 }, { "clip_ratio/high_max": 0.01464845088776201, "clip_ratio/high_mean": 0.01464845088776201, "clip_ratio/low_mean": 0.006000505993142724, "clip_ratio/low_min": 0.006000505993142724, "clip_ratio/region_mean": 0.020648956880904734, "completions/clipped_ratio": 0.0, "completions/max_length": 293.0, "completions/max_terminated_length": 293.0, "completions/mean_length": 290.375, "completions/mean_terminated_length": 290.375, "completions/min_length": 287.0, "completions/min_terminated_length": 287.0, "entropy": 0.24721034429967403, "epoch": 0.08924770052616782, "frac_reward_zero_std": 0.0, "grad_norm": 1.5890640020370483, "learning_rate": 3.2696969696969698e-06, "loss": 0.0058, "num_tokens": 5025785.0, "reward": 0.6659300327301025, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7272727489471436, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988961219787598, "reward_meter_std": 0.0002896705409511924, "reward_repeat_penalty_mean": 0.9166666269302368, "reward_repeat_penalty_std": 0.0471404492855072, "reward_std": 0.03423101827502251, "reward_total_composite_mean": 0.6659300327301025, "reward_total_composite_std": 0.0342310331761837, "reward_total_mean": 0.6659300327301025, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7272727489471436, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988961219787598, "rewards/meter/std": 0.0002896705409511924, "rewards/repeat_penalty/mean": 0.9166666269302368, "rewards/repeat_penalty/std": 0.0471404492855072, "rewards/total_composite/mean": 0.6659300327301025, "rewards/total_composite/std": 0.0342310331761837, "sampling/importance_sampling_ratio/max": 1.77455735206604, "sampling/importance_sampling_ratio/mean": 1.0035452842712402, "sampling/importance_sampling_ratio/min": 0.22851894795894623, "sampling/sampling_logp_difference/max": 1.4761362075805664, "sampling/sampling_logp_difference/mean": 0.02183903567492962, "step": 2222 }, { "clip_ratio/high_max": 0.002923976629972458, "clip_ratio/high_mean": 0.002923976629972458, "clip_ratio/low_mean": 0.005715410457924008, "clip_ratio/low_min": 0.005715410457924008, "clip_ratio/region_mean": 0.008639387087896466, "completions/clipped_ratio": 0.0, "completions/max_length": 182.0, "completions/max_terminated_length": 182.0, "completions/mean_length": 172.375, "completions/mean_terminated_length": 172.375, "completions/min_length": 171.0, "completions/min_terminated_length": 171.0, "entropy": 0.0778789222240448, "epoch": 0.08928786600795277, "frac_reward_zero_std": 0.0, "grad_norm": 1.8738999366760254, "learning_rate": 3.266666666666667e-06, "loss": 0.0074, "num_tokens": 5028644.0, "reward": 0.6164560317993164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9511035084724426, "reward_meter_std": 0.01162177324295044, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0075326296500861645, "reward_total_composite_mean": 0.6164560317993164, "reward_total_composite_std": 0.00753263384103775, "reward_total_mean": 0.6164560317993164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9511035084724426, "rewards/meter/std": 0.01162177324295044, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6164560317993164, "rewards/total_composite/std": 0.00753263384103775, "sampling/importance_sampling_ratio/max": 1.608762502670288, "sampling/importance_sampling_ratio/mean": 1.002518892288208, "sampling/importance_sampling_ratio/min": 0.5107390880584717, "sampling/sampling_logp_difference/max": 0.6718964576721191, "sampling/sampling_logp_difference/mean": 0.012546628713607788, "step": 2223 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.00872120977146551, "epoch": 0.08932803148973772, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.263636363636364e-06, "loss": 0.0, "num_tokens": 5030428.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0260026454925537, "sampling/importance_sampling_ratio/mean": 1.000982403755188, "sampling/importance_sampling_ratio/min": 0.9967485070228577, "sampling/sampling_logp_difference/max": 0.025670286267995834, "sampling/sampling_logp_difference/mean": 0.0009967123623937368, "step": 2224 }, { "clip_ratio/high_max": 0.021526333526708186, "clip_ratio/high_mean": 0.021526333526708186, "clip_ratio/low_mean": 0.0032748287776485085, "clip_ratio/low_min": 0.0032748287776485085, "clip_ratio/region_mean": 0.024801162304356694, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 76.5, "completions/mean_terminated_length": 76.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.27160678431391716, "epoch": 0.08936819697152268, "frac_reward_zero_std": 0.0, "grad_norm": 3.8838281631469727, "learning_rate": 3.2606060606060607e-06, "loss": 0.0108, "num_tokens": 5032200.0, "reward": 0.9987069964408875, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987069964408875, "reward_meter_std": 0.0007462438079528511, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007462415378540754, "reward_total_composite_mean": 0.9987069964408875, "reward_total_composite_std": 0.0007462438079528511, "reward_total_mean": 0.9987069964408875, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987069964408875, "rewards/meter/std": 0.0007462438079528511, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987069964408875, "rewards/total_composite/std": 0.0007462438079528511, "sampling/importance_sampling_ratio/max": 1.6162910461425781, "sampling/importance_sampling_ratio/mean": 1.0066579580307007, "sampling/importance_sampling_ratio/min": 0.37914493680000305, "sampling/sampling_logp_difference/max": 0.9698367118835449, "sampling/sampling_logp_difference/mean": 0.03162068501114845, "step": 2225 }, { "clip_ratio/high_max": 0.010416666744276881, "clip_ratio/high_mean": 0.010416666744276881, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.010416666744276881, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 36.375, "completions/mean_terminated_length": 36.375, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.18548811599612236, "epoch": 0.08940836245330763, "frac_reward_zero_std": 0.0, "grad_norm": 7.719726085662842, "learning_rate": 3.257575757575758e-06, "loss": 0.0153, "num_tokens": 5033803.0, "reward": 0.9948956370353699, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948956370353699, "reward_meter_std": 0.005730676930397749, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005730684380978346, "reward_total_composite_mean": 0.9948956370353699, "reward_total_composite_std": 0.005730676930397749, "reward_total_mean": 0.9948956370353699, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948956370353699, "rewards/meter/std": 0.005730676930397749, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948956370353699, "rewards/total_composite/std": 0.005730676930397749, "sampling/importance_sampling_ratio/max": 1.3892041444778442, "sampling/importance_sampling_ratio/mean": 1.006754994392395, "sampling/importance_sampling_ratio/min": 0.369167685508728, "sampling/sampling_logp_difference/max": 0.9965043067932129, "sampling/sampling_logp_difference/mean": 0.028130551800131798, "step": 2226 }, { "clip_ratio/high_max": 0.005880667129531503, "clip_ratio/high_mean": 0.005880667129531503, "clip_ratio/low_mean": 0.014457646873779595, "clip_ratio/low_min": 0.014457646873779595, "clip_ratio/region_mean": 0.020338314003311098, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 505.0, "completions/mean_length": 477.375, "completions/mean_terminated_length": 472.4285888671875, "completions/min_length": 445.0, "completions/min_terminated_length": 445.0, "entropy": 0.21459073200821877, "epoch": 0.08944852793509259, "frac_reward_zero_std": 0.0, "grad_norm": 1.227808952331543, "learning_rate": 3.2545454545454548e-06, "loss": 0.205, "num_tokens": 5038942.0, "reward": 0.4574677348136902, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5625, "reward_count_adherence_std": 0.0353553481400013, "reward_meter_mean": 0.9960645437240601, "reward_meter_std": 0.0029486457351595163, "reward_repeat_penalty_mean": 0.8150820732116699, "reward_repeat_penalty_std": 0.08796744048595428, "reward_std": 0.06483428180217743, "reward_total_composite_mean": 0.4574677348136902, "reward_total_composite_std": 0.06483428180217743, "reward_total_mean": 0.4574677348136902, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5625, "rewards/count_adherence/std": 0.0353553481400013, "rewards/meter/mean": 0.9960645437240601, "rewards/meter/std": 0.0029486457351595163, "rewards/repeat_penalty/mean": 0.8150820732116699, "rewards/repeat_penalty/std": 0.08796744048595428, "rewards/total_composite/mean": 0.4574677348136902, "rewards/total_composite/std": 0.06483428180217743, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0072216987609863, "sampling/importance_sampling_ratio/min": 0.07101205736398697, "sampling/sampling_logp_difference/max": 2.6449055671691895, "sampling/sampling_logp_difference/mean": 0.03101653791964054, "step": 2227 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.007959658920299262, "epoch": 0.08948869341687754, "frac_reward_zero_std": 0.0, "grad_norm": 0.01318140048533678, "learning_rate": 3.2515151515151516e-06, "loss": 0.0001, "num_tokens": 5040574.0, "reward": 0.7876332998275757, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7876332998275757, "reward_meter_std": 1.5573279597447254e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.558229632792063e-05, "reward_total_composite_mean": 0.7876332998275757, "reward_total_composite_std": 1.5573279597447254e-05, "reward_total_mean": 0.7876332998275757, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7876332998275757, "rewards/meter/std": 1.5573279597447254e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876332998275757, "rewards/total_composite/std": 1.5573279597447254e-05, "sampling/importance_sampling_ratio/max": 1.068850040435791, "sampling/importance_sampling_ratio/mean": 1.0008612871170044, "sampling/importance_sampling_ratio/min": 0.991096556186676, "sampling/sampling_logp_difference/max": 0.06658339500427246, "sampling/sampling_logp_difference/mean": 0.0009326103026978672, "step": 2228 }, { "clip_ratio/high_max": 0.010519623290747404, "clip_ratio/high_mean": 0.010519623290747404, "clip_ratio/low_mean": 0.0015151514671742916, "clip_ratio/low_min": 0.0015151514671742916, "clip_ratio/region_mean": 0.012034774757921696, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 166.0, "completions/mean_terminated_length": 166.0, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.06033759843558073, "epoch": 0.0895288588986625, "frac_reward_zero_std": 0.0, "grad_norm": 2.3057451248168945, "learning_rate": 3.2484848484848484e-06, "loss": -0.0005, "num_tokens": 5043518.0, "reward": 0.7165045738220215, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984830617904663, "reward_meter_std": 6.16059624007903e-05, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.05143444612622261, "reward_std": 0.042806532233953476, "reward_total_composite_mean": 0.7165045738220215, "reward_total_composite_std": 0.042806532233953476, "reward_total_mean": 0.7165045738220215, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984830617904663, "rewards/meter/std": 6.16059624007903e-05, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.7165045738220215, "rewards/total_composite/std": 0.042806532233953476, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0007810592651367, "sampling/importance_sampling_ratio/min": 0.0514800027012825, "sampling/sampling_logp_difference/max": 2.966561794281006, "sampling/sampling_logp_difference/mean": 0.017393097281455994, "step": 2229 }, { "clip_ratio/high_max": 0.02239304850809276, "clip_ratio/high_mean": 0.02239304850809276, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.02239304850809276, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 33.875, "completions/mean_terminated_length": 33.875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.1314310822635889, "epoch": 0.08956902438044745, "frac_reward_zero_std": 0.0, "grad_norm": 10.35647964477539, "learning_rate": 3.2454545454545457e-06, "loss": 0.0099, "num_tokens": 5044997.0, "reward": 0.9669080972671509, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9669080972671509, "reward_meter_std": 0.007700720801949501, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007700727321207523, "reward_total_composite_mean": 0.9669080972671509, "reward_total_composite_std": 0.007700720801949501, "reward_total_mean": 0.9669080972671509, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9669080972671509, "rewards/meter/std": 0.007700720801949501, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9669080972671509, "rewards/total_composite/std": 0.007700720801949501, "sampling/importance_sampling_ratio/max": 1.6818870306015015, "sampling/importance_sampling_ratio/mean": 1.003213882446289, "sampling/importance_sampling_ratio/min": 0.07987331598997116, "sampling/sampling_logp_difference/max": 2.527313470840454, "sampling/sampling_logp_difference/mean": 0.042054299265146255, "step": 2230 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.010291360202245414, "epoch": 0.0896091898622324, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.2424242424242425e-06, "loss": 0.0, "num_tokens": 5046893.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.023606777191162, "sampling/importance_sampling_ratio/mean": 1.0010844469070435, "sampling/importance_sampling_ratio/min": 0.9904347658157349, "sampling/sampling_logp_difference/max": 0.02333247661590576, "sampling/sampling_logp_difference/mean": 0.0011728927493095398, "step": 2231 }, { "clip_ratio/high_max": 0.0078125, "clip_ratio/high_mean": 0.0078125, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0078125, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.019377628806978464, "epoch": 0.08964935534401736, "frac_reward_zero_std": 0.0, "grad_norm": 0.10460558533668518, "learning_rate": 3.2393939393939393e-06, "loss": -0.0002, "num_tokens": 5048693.0, "reward": 0.9992489814758301, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992489814758301, "reward_meter_std": 1.6926040188991465e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.6926040188991465e-05, "reward_total_composite_mean": 0.9992489814758301, "reward_total_composite_std": 1.6926040188991465e-05, "reward_total_mean": 0.9992489814758301, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992489814758301, "rewards/meter/std": 1.6926040188991465e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992489814758301, "rewards/total_composite/std": 1.6926040188991465e-05, "sampling/importance_sampling_ratio/max": 1.2030258178710938, "sampling/importance_sampling_ratio/mean": 0.9996911883354187, "sampling/importance_sampling_ratio/min": 0.24956819415092468, "sampling/sampling_logp_difference/max": 1.3880231380462646, "sampling/sampling_logp_difference/mean": 0.006007255986332893, "step": 2232 }, { "clip_ratio/high_max": 0.005999534041620791, "clip_ratio/high_mean": 0.005999534041620791, "clip_ratio/low_mean": 0.012797378934919834, "clip_ratio/low_min": 0.012797378934919834, "clip_ratio/region_mean": 0.018796912976540625, "completions/clipped_ratio": 0.0, "completions/max_length": 149.0, "completions/max_terminated_length": 149.0, "completions/mean_length": 145.875, "completions/mean_terminated_length": 145.875, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.2233992125838995, "epoch": 0.08968952082580231, "frac_reward_zero_std": 0.0, "grad_norm": 1.9100828170776367, "learning_rate": 3.236363636363636e-06, "loss": 0.0026, "num_tokens": 5051460.0, "reward": 0.7277184724807739, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988158941268921, "reward_meter_std": 0.0003107816446572542, "reward_repeat_penalty_mean": 0.9107142686843872, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.059221841394901276, "reward_total_composite_mean": 0.7277184724807739, "reward_total_composite_std": 0.059221845120191574, "reward_total_mean": 0.7277184724807739, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988158941268921, "rewards/meter/std": 0.0003107816446572542, "rewards/repeat_penalty/mean": 0.9107142686843872, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.7277184724807739, "rewards/total_composite/std": 0.059221845120191574, "sampling/importance_sampling_ratio/max": 1.9528251886367798, "sampling/importance_sampling_ratio/mean": 1.0000747442245483, "sampling/importance_sampling_ratio/min": 0.18826662003993988, "sampling/sampling_logp_difference/max": 1.669896125793457, "sampling/sampling_logp_difference/mean": 0.027437884360551834, "step": 2233 }, { "clip_ratio/high_max": 0.017986543010920286, "clip_ratio/high_mean": 0.017986543010920286, "clip_ratio/low_mean": 0.005488064838573337, "clip_ratio/low_min": 0.005488064838573337, "clip_ratio/region_mean": 0.023474607849493623, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.75, "completions/mean_terminated_length": 68.75, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.10225279349833727, "epoch": 0.08972968630758726, "frac_reward_zero_std": 0.0, "grad_norm": 3.689906120300293, "learning_rate": 3.2333333333333334e-06, "loss": 0.0004, "num_tokens": 5053322.0, "reward": 0.9638321399688721, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9638321399688721, "reward_meter_std": 0.004275594372302294, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004275602288544178, "reward_total_composite_mean": 0.9638321399688721, "reward_total_composite_std": 0.004275594372302294, "reward_total_mean": 0.9638321399688721, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9638321399688721, "rewards/meter/std": 0.004275594372302294, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9638321399688721, "rewards/total_composite/std": 0.004275594372302294, "sampling/importance_sampling_ratio/max": 1.3294564485549927, "sampling/importance_sampling_ratio/mean": 0.9958019852638245, "sampling/importance_sampling_ratio/min": 0.2931382656097412, "sampling/sampling_logp_difference/max": 1.2271108627319336, "sampling/sampling_logp_difference/mean": 0.02401396632194519, "step": 2234 }, { "clip_ratio/high_max": 0.005434782709926367, "clip_ratio/high_mean": 0.005434782709926367, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/region_mean": 0.007220497005619109, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.06750913895666599, "epoch": 0.08976985178937222, "frac_reward_zero_std": 0.0, "grad_norm": 1.4711960554122925, "learning_rate": 3.2303030303030307e-06, "loss": 0.0039, "num_tokens": 5055211.0, "reward": 0.9970292448997498, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970292448997498, "reward_meter_std": 0.0009962875628843904, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009962901240214705, "reward_total_composite_mean": 0.9970292448997498, "reward_total_composite_std": 0.0009962875628843904, "reward_total_mean": 0.9970292448997498, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970292448997498, "rewards/meter/std": 0.0009962875628843904, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970292448997498, "rewards/total_composite/std": 0.0009962875628843904, "sampling/importance_sampling_ratio/max": 1.4033783674240112, "sampling/importance_sampling_ratio/mean": 1.0035878419876099, "sampling/importance_sampling_ratio/min": 0.6541445255279541, "sampling/sampling_logp_difference/max": 0.4244270324707031, "sampling/sampling_logp_difference/mean": 0.008085141889750957, "step": 2235 }, { "clip_ratio/high_max": 0.006067961221560836, "clip_ratio/high_mean": 0.006067961221560836, "clip_ratio/low_mean": 0.0036407767329365015, "clip_ratio/low_min": 0.0036407767329365015, "clip_ratio/region_mean": 0.009708737954497337, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 102.75, "completions/mean_terminated_length": 102.75, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.05867503536865115, "epoch": 0.08981001727115717, "frac_reward_zero_std": 0.0, "grad_norm": 0.5810080766677856, "learning_rate": 3.227272727272728e-06, "loss": 0.0014, "num_tokens": 5057433.0, "reward": 0.9973079562187195, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973079562187195, "reward_meter_std": 4.1739614971447736e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.174207424512133e-05, "reward_total_composite_mean": 0.9973079562187195, "reward_total_composite_std": 4.1739614971447736e-05, "reward_total_mean": 0.9973079562187195, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973079562187195, "rewards/meter/std": 4.1739614971447736e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973079562187195, "rewards/total_composite/std": 4.1739614971447736e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0025743246078491, "sampling/importance_sampling_ratio/min": 0.39697858691215515, "sampling/sampling_logp_difference/max": 0.9729032516479492, "sampling/sampling_logp_difference/mean": 0.008572143502533436, "step": 2236 }, { "clip_ratio/high_max": 0.021151655237190425, "clip_ratio/high_mean": 0.021151655237190425, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/region_mean": 0.025959347723983228, "completions/clipped_ratio": 0.0, "completions/max_length": 164.0, "completions/max_terminated_length": 164.0, "completions/mean_length": 158.375, "completions/mean_terminated_length": 158.375, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.28765890188515186, "epoch": 0.08985018275294213, "frac_reward_zero_std": 0.0, "grad_norm": 2.2996366024017334, "learning_rate": 3.2242424242424248e-06, "loss": -0.0042, "num_tokens": 5060180.0, "reward": 0.7824714183807373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958896636962891, "reward_meter_std": 0.0022456336300820112, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.04001671448349953, "reward_total_composite_mean": 0.7824714183807373, "reward_total_composite_std": 0.040016695857048035, "reward_total_mean": 0.7824714183807373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958896636962891, "rewards/meter/std": 0.0022456336300820112, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.7824714183807373, "rewards/total_composite/std": 0.040016695857048035, "sampling/importance_sampling_ratio/max": 1.8041925430297852, "sampling/importance_sampling_ratio/mean": 1.003827452659607, "sampling/importance_sampling_ratio/min": 0.3160504102706909, "sampling/sampling_logp_difference/max": 1.1518535614013672, "sampling/sampling_logp_difference/mean": 0.0335751473903656, "step": 2237 }, { "clip_ratio/high_max": 0.010869565419852734, "clip_ratio/high_mean": 0.010869565419852734, "clip_ratio/low_mean": 0.005461423774249852, "clip_ratio/low_min": 0.005461423774249852, "clip_ratio/region_mean": 0.016330989194102585, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.875, "completions/mean_terminated_length": 68.875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.07339224359020591, "epoch": 0.08989034823472708, "frac_reward_zero_std": 0.0, "grad_norm": 0.5047301650047302, "learning_rate": 3.2212121212121216e-06, "loss": 0.0, "num_tokens": 5061987.0, "reward": 0.9973945617675781, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973945617675781, "reward_meter_std": 5.974572559352964e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.973832230665721e-05, "reward_total_composite_mean": 0.9973945617675781, "reward_total_composite_std": 5.974572559352964e-05, "reward_total_mean": 0.9973945617675781, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973945617675781, "rewards/meter/std": 5.974572559352964e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973945617675781, "rewards/total_composite/std": 5.974572559352964e-05, "sampling/importance_sampling_ratio/max": 1.4584788084030151, "sampling/importance_sampling_ratio/mean": 1.0013201236724854, "sampling/importance_sampling_ratio/min": 0.7254200577735901, "sampling/sampling_logp_difference/max": 0.3773939609527588, "sampling/sampling_logp_difference/mean": 0.008767606690526009, "step": 2238 }, { "clip_ratio/high_max": 0.00717289699241519, "clip_ratio/high_mean": 0.00717289699241519, "clip_ratio/low_mean": 0.010472985217347741, "clip_ratio/low_min": 0.010472985217347741, "clip_ratio/region_mean": 0.01764588220976293, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 106.0, "completions/mean_terminated_length": 106.0, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.1653594933450222, "epoch": 0.08993051371651203, "frac_reward_zero_std": 0.0, "grad_norm": 4.3548431396484375, "learning_rate": 3.2181818181818184e-06, "loss": 0.0221, "num_tokens": 5064107.0, "reward": 0.8447941541671753, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9938052296638489, "reward_meter_std": 0.002556393388658762, "reward_repeat_penalty_mean": 0.8500000238418579, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09272917360067368, "reward_total_composite_mean": 0.8447941541671753, "reward_total_composite_std": 0.09272917360067368, "reward_total_mean": 0.8447941541671753, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9938052296638489, "rewards/meter/std": 0.002556393388658762, "rewards/repeat_penalty/mean": 0.8500000238418579, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.8447941541671753, "rewards/total_composite/std": 0.09272917360067368, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017590522766113, "sampling/importance_sampling_ratio/min": 0.3138589859008789, "sampling/sampling_logp_difference/max": 1.215738296508789, "sampling/sampling_logp_difference/mean": 0.02209414355456829, "step": 2239 }, { "clip_ratio/high_max": 0.036260453751310706, "clip_ratio/high_mean": 0.036260453751310706, "clip_ratio/low_mean": 0.008410736452788115, "clip_ratio/low_min": 0.008410736452788115, "clip_ratio/region_mean": 0.04467119020409882, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 112.375, "completions/mean_terminated_length": 112.375, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.3458435367792845, "epoch": 0.08997067919829699, "frac_reward_zero_std": 0.0, "grad_norm": 6.652218341827393, "learning_rate": 3.2151515151515157e-06, "loss": 0.0227, "num_tokens": 5066446.0, "reward": 0.8900635242462158, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8900635242462158, "reward_meter_std": 0.201475590467453, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2014755755662918, "reward_total_composite_mean": 0.8900635242462158, "reward_total_composite_std": 0.201475590467453, "reward_total_mean": 0.8900635242462158, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8900635242462158, "rewards/meter/std": 0.201475590467453, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8900635242462158, "rewards/total_composite/std": 0.201475590467453, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028852224349976, "sampling/importance_sampling_ratio/min": 0.19080470502376556, "sampling/sampling_logp_difference/max": 1.6565048694610596, "sampling/sampling_logp_difference/mean": 0.05001511052250862, "step": 2240 }, { "clip_ratio/high_max": 0.01353727001696825, "clip_ratio/high_mean": 0.01353727001696825, "clip_ratio/low_mean": 0.006711711874231696, "clip_ratio/low_min": 0.006711711874231696, "clip_ratio/region_mean": 0.020248981891199946, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 74.125, "completions/mean_terminated_length": 74.125, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.2299621980637312, "epoch": 0.09001084468008194, "frac_reward_zero_std": 0.0, "grad_norm": 2.789885997772217, "learning_rate": 3.2121212121212125e-06, "loss": -0.0017, "num_tokens": 5068415.0, "reward": 0.9991567134857178, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991567134857178, "reward_meter_std": 0.0002589829673524946, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00025898346211761236, "reward_total_composite_mean": 0.9991567134857178, "reward_total_composite_std": 0.0002589829673524946, "reward_total_mean": 0.9991567134857178, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991567134857178, "rewards/meter/std": 0.0002589829673524946, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991567134857178, "rewards/total_composite/std": 0.0002589829673524946, "sampling/importance_sampling_ratio/max": 1.596889853477478, "sampling/importance_sampling_ratio/mean": 1.0056432485580444, "sampling/importance_sampling_ratio/min": 0.34301483631134033, "sampling/sampling_logp_difference/max": 1.069981575012207, "sampling/sampling_logp_difference/mean": 0.029841933399438858, "step": 2241 }, { "clip_ratio/high_max": 0.007246376946568489, "clip_ratio/high_mean": 0.007246376946568489, "clip_ratio/low_mean": 0.007326300139538944, "clip_ratio/low_min": 0.007326300139538944, "clip_ratio/region_mean": 0.014572677086107433, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.75, "completions/mean_terminated_length": 68.75, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.06746953492984176, "epoch": 0.0900510101618669, "frac_reward_zero_std": 0.0, "grad_norm": 1.4458215236663818, "learning_rate": 3.2090909090909094e-06, "loss": -0.0007, "num_tokens": 5070277.0, "reward": 0.9974172115325928, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974172115325928, "reward_meter_std": 9.149048128165305e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.149555262411013e-05, "reward_total_composite_mean": 0.9974172115325928, "reward_total_composite_std": 9.149048128165305e-05, "reward_total_mean": 0.9974172115325928, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974172115325928, "rewards/meter/std": 9.149048128165305e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974172115325928, "rewards/total_composite/std": 9.149048128165305e-05, "sampling/importance_sampling_ratio/max": 1.4807547330856323, "sampling/importance_sampling_ratio/mean": 1.0000346899032593, "sampling/importance_sampling_ratio/min": 0.3756442964076996, "sampling/sampling_logp_difference/max": 0.9791126251220703, "sampling/sampling_logp_difference/mean": 0.01175268180668354, "step": 2242 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 55.0, "completions/mean_terminated_length": 55.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.01730545365717262, "epoch": 0.09009117564365185, "frac_reward_zero_std": 0.0, "grad_norm": 0.028668954968452454, "learning_rate": 3.2060606060606066e-06, "loss": -0.0004, "num_tokens": 5071893.0, "reward": 0.9948811531066895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948811531066895, "reward_meter_std": 1.4479670653599896e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.448755483579589e-06, "reward_total_composite_mean": 0.9948811531066895, "reward_total_composite_std": 1.4479670653599896e-06, "reward_total_mean": 0.9948811531066895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948811531066895, "rewards/meter/std": 1.4479670653599896e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948811531066895, "rewards/total_composite/std": 1.4479670653599896e-06, "sampling/importance_sampling_ratio/max": 1.0825896263122559, "sampling/importance_sampling_ratio/mean": 0.9990595579147339, "sampling/importance_sampling_ratio/min": 0.45439019799232483, "sampling/sampling_logp_difference/max": 0.7887990474700928, "sampling/sampling_logp_difference/mean": 0.005395011510699987, "step": 2243 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.013456470100209117, "clip_ratio/low_min": 0.013456470100209117, "clip_ratio/region_mean": 0.015294705401174724, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 70.5, "completions/mean_terminated_length": 70.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.14800493232905865, "epoch": 0.0901313411254368, "frac_reward_zero_std": 0.0, "grad_norm": 2.5679850578308105, "learning_rate": 3.2030303030303034e-06, "loss": 0.0324, "num_tokens": 5073801.0, "reward": 0.9620155096054077, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9620155096054077, "reward_meter_std": 0.010956778191030025, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010956759564578533, "reward_total_composite_mean": 0.9620155096054077, "reward_total_composite_std": 0.010956778191030025, "reward_total_mean": 0.9620155096054077, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9620155096054077, "rewards/meter/std": 0.010956778191030025, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9620155096054077, "rewards/total_composite/std": 0.010956778191030025, "sampling/importance_sampling_ratio/max": 1.9989465475082397, "sampling/importance_sampling_ratio/mean": 1.0068614482879639, "sampling/importance_sampling_ratio/min": 0.19815008342266083, "sampling/sampling_logp_difference/max": 1.6187305450439453, "sampling/sampling_logp_difference/mean": 0.026699479669332504, "step": 2244 }, { "clip_ratio/high_max": 0.005239521153271198, "clip_ratio/high_mean": 0.005239521153271198, "clip_ratio/low_mean": 0.0022455090656876564, "clip_ratio/low_min": 0.0022455090656876564, "clip_ratio/region_mean": 0.007485030218958855, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 167.0, "completions/mean_terminated_length": 167.0, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "entropy": 0.05933140218257904, "epoch": 0.09017150660722176, "frac_reward_zero_std": 0.0, "grad_norm": 1.3484723567962646, "learning_rate": 3.2000000000000003e-06, "loss": 0.0006, "num_tokens": 5076689.0, "reward": 0.717853307723999, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959465265274048, "reward_meter_std": 9.931746171787381e-05, "reward_repeat_penalty_mean": 0.8409091234207153, "reward_repeat_penalty_std": 0.08058230578899384, "reward_std": 0.06874334812164307, "reward_total_composite_mean": 0.717853307723999, "reward_total_composite_std": 0.06874335557222366, "reward_total_mean": 0.717853307723999, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959465265274048, "rewards/meter/std": 9.931746171787381e-05, "rewards/repeat_penalty/mean": 0.8409091234207153, "rewards/repeat_penalty/std": 0.08058230578899384, "rewards/total_composite/mean": 0.717853307723999, "rewards/total_composite/std": 0.06874335557222366, "sampling/importance_sampling_ratio/max": 1.3406785726547241, "sampling/importance_sampling_ratio/mean": 0.9994251132011414, "sampling/importance_sampling_ratio/min": 0.29894891381263733, "sampling/sampling_logp_difference/max": 1.2074825763702393, "sampling/sampling_logp_difference/mean": 0.010420297272503376, "step": 2245 }, { "clip_ratio/high_max": 0.018073388375341892, "clip_ratio/high_mean": 0.018073388375341892, "clip_ratio/low_mean": 0.0008802816737443209, "clip_ratio/low_min": 0.0008802816737443209, "clip_ratio/region_mean": 0.018953670049086213, "completions/clipped_ratio": 0.0, "completions/max_length": 164.0, "completions/max_terminated_length": 164.0, "completions/mean_length": 143.5, "completions/mean_terminated_length": 143.5, "completions/min_length": 136.0, "completions/min_terminated_length": 136.0, "entropy": 0.19038555212318897, "epoch": 0.09021167208900671, "frac_reward_zero_std": 0.0, "grad_norm": 2.09184193611145, "learning_rate": 3.196969696969697e-06, "loss": 0.0044, "num_tokens": 5079477.0, "reward": 0.6652302742004395, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9908416867256165, "reward_meter_std": 0.005638771690428257, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.03939947113394737, "reward_total_composite_mean": 0.6652302742004395, "reward_total_composite_std": 0.03939948230981827, "reward_total_mean": 0.6652302742004395, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9908416867256165, "rewards/meter/std": 0.005638771690428257, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.6652302742004395, "rewards/total_composite/std": 0.03939948230981827, "sampling/importance_sampling_ratio/max": 1.8299243450164795, "sampling/importance_sampling_ratio/mean": 1.004660725593567, "sampling/importance_sampling_ratio/min": 0.22255297005176544, "sampling/sampling_logp_difference/max": 1.5025901794433594, "sampling/sampling_logp_difference/mean": 0.02601451985538006, "step": 2246 }, { "clip_ratio/high_max": 0.0008741258643567562, "clip_ratio/high_mean": 0.0008741258643567562, "clip_ratio/low_mean": 0.001760650658980012, "clip_ratio/low_min": 0.001760650658980012, "clip_ratio/region_mean": 0.002634776523336768, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 142.625, "completions/mean_terminated_length": 142.625, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 0.10466479323804379, "epoch": 0.09025183757079167, "frac_reward_zero_std": 0.0, "grad_norm": 2.32881760597229, "learning_rate": 3.1939393939393944e-06, "loss": 0.0006, "num_tokens": 5082066.0, "reward": 0.8519160747528076, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939021468162537, "reward_meter_std": 0.0005415793275460601, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004642065614461899, "reward_total_composite_mean": 0.8519160747528076, "reward_total_composite_std": 0.0004642106650862843, "reward_total_mean": 0.8519160747528076, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939021468162537, "rewards/meter/std": 0.0005415793275460601, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8519160747528076, "rewards/total_composite/std": 0.0004642106650862843, "sampling/importance_sampling_ratio/max": 1.7263494729995728, "sampling/importance_sampling_ratio/mean": 1.001724362373352, "sampling/importance_sampling_ratio/min": 0.2304760217666626, "sampling/sampling_logp_difference/max": 1.4676084518432617, "sampling/sampling_logp_difference/mean": 0.014642242342233658, "step": 2247 }, { "clip_ratio/high_max": 0.0045045046135783195, "clip_ratio/high_mean": 0.0045045046135783195, "clip_ratio/low_mean": 0.0022522523067891598, "clip_ratio/low_min": 0.0022522523067891598, "clip_ratio/region_mean": 0.006756756920367479, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 111.0, "completions/mean_terminated_length": 111.0, "completions/min_length": 111.0, "completions/min_terminated_length": 111.0, "entropy": 0.06026614410802722, "epoch": 0.09029200305257662, "frac_reward_zero_std": 0.0, "grad_norm": 2.4443159103393555, "learning_rate": 3.190909090909091e-06, "loss": 0.0011, "num_tokens": 5084426.0, "reward": 0.9422965049743652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956398010253906, "reward_meter_std": 9.625943494029343e-05, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07353662699460983, "reward_total_composite_mean": 0.9422965049743652, "reward_total_composite_std": 0.07353661954402924, "reward_total_mean": 0.9422965049743652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956398010253906, "rewards/meter/std": 9.625943494029343e-05, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9422965049743652, "rewards/total_composite/std": 0.07353661954402924, "sampling/importance_sampling_ratio/max": 1.3495516777038574, "sampling/importance_sampling_ratio/mean": 1.0003148317337036, "sampling/importance_sampling_ratio/min": 0.5978375673294067, "sampling/sampling_logp_difference/max": 0.5144362449645996, "sampling/sampling_logp_difference/mean": 0.008148727007210255, "step": 2248 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.05507086403667927, "epoch": 0.09033216853436157, "frac_reward_zero_std": 0.0, "grad_norm": 0.3145919442176819, "learning_rate": 3.187878787878788e-06, "loss": 0.0001, "num_tokens": 5086546.0, "reward": 0.9975405931472778, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975405931472778, "reward_meter_std": 1.4270765859691892e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.4251797438191716e-05, "reward_total_composite_mean": 0.9975405931472778, "reward_total_composite_std": 1.4270765859691892e-05, "reward_total_mean": 0.9975405931472778, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975405931472778, "rewards/meter/std": 1.4270765859691892e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975405931472778, "rewards/total_composite/std": 1.4270765859691892e-05, "sampling/importance_sampling_ratio/max": 1.2142425775527954, "sampling/importance_sampling_ratio/mean": 1.0008234977722168, "sampling/importance_sampling_ratio/min": 0.5022322535514832, "sampling/sampling_logp_difference/max": 0.688692569732666, "sampling/sampling_logp_difference/mean": 0.006101167760789394, "step": 2249 }, { "clip_ratio/high_max": 0.0036231884732842445, "clip_ratio/high_mean": 0.0036231884732842445, "clip_ratio/low_mean": 0.0036498295376077294, "clip_ratio/low_min": 0.0036498295376077294, "clip_ratio/region_mean": 0.007273018010891974, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.875, "completions/mean_terminated_length": 68.875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.0506328740157187, "epoch": 0.09037233401614653, "frac_reward_zero_std": 0.0, "grad_norm": 0.5331827402114868, "learning_rate": 3.1848484848484853e-06, "loss": -0.0011, "num_tokens": 5088505.0, "reward": 0.9975307583808899, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975307583808899, "reward_meter_std": 2.9643033485626802e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.9639279091497883e-05, "reward_total_composite_mean": 0.9975307583808899, "reward_total_composite_std": 2.9643033485626802e-05, "reward_total_mean": 0.9975307583808899, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975307583808899, "rewards/meter/std": 2.9643033485626802e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975307583808899, "rewards/total_composite/std": 2.9643033485626802e-05, "sampling/importance_sampling_ratio/max": 1.2806227207183838, "sampling/importance_sampling_ratio/mean": 1.0018889904022217, "sampling/importance_sampling_ratio/min": 0.5288791656494141, "sampling/sampling_logp_difference/max": 0.6369953155517578, "sampling/sampling_logp_difference/mean": 0.005882658995687962, "step": 2250 }, { "epoch": 0.09037233401614653, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/max_length": 330.53846153846155, "eval_completions/max_terminated_length": 319.53846153846155, "eval_completions/mean_length": 182.43269230769232, "eval_completions/mean_terminated_length": 180.01648418719952, "eval_completions/min_length": 59.53846153846154, "eval_completions/min_terminated_length": 59.53846153846154, "eval_entropy": 0.19319924941429725, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 5088505.0, "eval_reward": 0.5561339992743272, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.8588019150954026, "eval_reward_count_adherence_std": 0.1499052156622593, "eval_reward_meter_mean": 0.732613939505357, "eval_reward_meter_std": 0.3826414684836681, "eval_reward_repeat_penalty_mean": 0.875946778517503, "eval_reward_repeat_penalty_std": 0.12293276706567177, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5561339992743272, "eval_reward_total_composite_std": 0.3474527712051685, "eval_reward_total_mean": 0.5561339992743272, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.8588019150954026, "eval_rewards/count_adherence/std": 0.1499052156622593, "eval_rewards/meter/mean": 0.732613939505357, "eval_rewards/meter/std": 0.3826414684836681, "eval_rewards/repeat_penalty/mean": 0.875946778517503, "eval_rewards/repeat_penalty/std": 0.12293276706567177, "eval_rewards/total_composite/mean": 0.5561339992743272, "eval_rewards/total_composite/std": 0.3474527712051685, "eval_runtime": 64.292, "eval_samples_per_second": 1.618, "eval_sampling/importance_sampling_ratio/max": 1.410868342106159, "eval_sampling/importance_sampling_ratio/mean": 1.0042996865052443, "eval_sampling/importance_sampling_ratio/min": 0.40484776290563435, "eval_sampling/sampling_logp_difference/max": 0.9445312756758469, "eval_sampling/sampling_logp_difference/mean": 0.0185515835451392, "eval_steps_per_second": 0.202, "step": 2250 }, { "clip_ratio/high_max": 0.003688839729875326, "clip_ratio/high_mean": 0.003688839729875326, "clip_ratio/low_mean": 0.009804157423786819, "clip_ratio/low_min": 0.009804157423786819, "clip_ratio/region_mean": 0.013492997153662145, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 102.0, "completions/mean_terminated_length": 102.0, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.15675346460193396, "epoch": 0.09041249949793148, "frac_reward_zero_std": 0.0, "grad_norm": 2.5881564617156982, "learning_rate": 3.181818181818182e-06, "loss": 0.0069, "num_tokens": 5090737.0, "reward": 0.9657948613166809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9657948613166809, "reward_meter_std": 0.005292544141411781, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005292540416121483, "reward_total_composite_mean": 0.9657948613166809, "reward_total_composite_std": 0.005292544141411781, "reward_total_mean": 0.9657948613166809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9657948613166809, "rewards/meter/std": 0.005292544141411781, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9657948613166809, "rewards/total_composite/std": 0.005292544141411781, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007267713546753, "sampling/importance_sampling_ratio/min": 0.32930082082748413, "sampling/sampling_logp_difference/max": 1.110783576965332, "sampling/sampling_logp_difference/mean": 0.022372471168637276, "step": 2251 }, { "clip_ratio/high_max": 0.031607592245563865, "clip_ratio/high_mean": 0.031607592245563865, "clip_ratio/low_mean": 0.012399396393448114, "clip_ratio/low_min": 0.012399396393448114, "clip_ratio/region_mean": 0.04400698863901198, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 71.0, "completions/mean_terminated_length": 71.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.43995901197195053, "epoch": 0.09045266497971644, "frac_reward_zero_std": 0.0, "grad_norm": 8.668261528015137, "learning_rate": 3.178787878787879e-06, "loss": 0.0092, "num_tokens": 5092817.0, "reward": 0.9977865219116211, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977865219116211, "reward_meter_std": 0.0025578902568668127, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0025578862987458706, "reward_total_composite_mean": 0.9977865219116211, "reward_total_composite_std": 0.0025578902568668127, "reward_total_mean": 0.9977865219116211, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977865219116211, "rewards/meter/std": 0.0025578902568668127, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977865219116211, "rewards/total_composite/std": 0.0025578902568668127, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0122182369232178, "sampling/importance_sampling_ratio/min": 0.2101971060037613, "sampling/sampling_logp_difference/max": 1.5597095489501953, "sampling/sampling_logp_difference/mean": 0.06398905813694, "step": 2252 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.01149469253141433, "epoch": 0.09049283046150139, "frac_reward_zero_std": 0.0, "grad_norm": 0.0889202132821083, "learning_rate": 3.1757575757575758e-06, "loss": -0.0002, "num_tokens": 5094681.0, "reward": 0.7876288890838623, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7876288890838623, "reward_meter_std": 2.8027659936924465e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.8021637263009325e-05, "reward_total_composite_mean": 0.7876288890838623, "reward_total_composite_std": 2.8027659936924465e-05, "reward_total_mean": 0.7876288890838623, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7876288890838623, "rewards/meter/std": 2.8027659936924465e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876288890838623, "rewards/total_composite/std": 2.8027659936924465e-05, "sampling/importance_sampling_ratio/max": 1.052701711654663, "sampling/importance_sampling_ratio/mean": 1.0006846189498901, "sampling/importance_sampling_ratio/min": 0.9315734505653381, "sampling/sampling_logp_difference/max": 0.07088017463684082, "sampling/sampling_logp_difference/mean": 0.0013179951347410679, "step": 2253 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.010870376718230546, "epoch": 0.09053299594328634, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.172727272727273e-06, "loss": 0.0, "num_tokens": 5096369.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.057936429977417, "sampling/importance_sampling_ratio/mean": 1.0007966756820679, "sampling/importance_sampling_ratio/min": 0.8877354264259338, "sampling/sampling_logp_difference/max": 0.11908148974180222, "sampling/sampling_logp_difference/mean": 0.0015978036681190133, "step": 2254 }, { "clip_ratio/high_max": 0.018939394503831863, "clip_ratio/high_mean": 0.018939394503831863, "clip_ratio/low_mean": 0.007464349502697587, "clip_ratio/low_min": 0.007464349502697587, "clip_ratio/region_mean": 0.02640374400652945, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.375, "completions/mean_terminated_length": 33.375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.1904709991067648, "epoch": 0.0905731614250713, "frac_reward_zero_std": 0.0, "grad_norm": 11.48292350769043, "learning_rate": 3.16969696969697e-06, "loss": -0.0042, "num_tokens": 5097900.0, "reward": 0.9692210555076599, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9692210555076599, "reward_meter_std": 0.003493846394121647, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0034938445314764977, "reward_total_composite_mean": 0.9692210555076599, "reward_total_composite_std": 0.003493846394121647, "reward_total_mean": 0.9692210555076599, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9692210555076599, "rewards/meter/std": 0.003493846394121647, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9692210555076599, "rewards/total_composite/std": 0.003493846394121647, "sampling/importance_sampling_ratio/max": 1.707613468170166, "sampling/importance_sampling_ratio/mean": 1.0003447532653809, "sampling/importance_sampling_ratio/min": 0.1905699372291565, "sampling/sampling_logp_difference/max": 1.65773606300354, "sampling/sampling_logp_difference/mean": 0.03685401380062103, "step": 2255 }, { "clip_ratio/high_max": 0.011609932873398066, "clip_ratio/high_mean": 0.011609932873398066, "clip_ratio/low_mean": 0.02541261224541813, "clip_ratio/low_min": 0.02541261224541813, "clip_ratio/region_mean": 0.0370225451188162, "completions/clipped_ratio": 0.0, "completions/max_length": 188.0, "completions/max_terminated_length": 188.0, "completions/mean_length": 176.5, "completions/mean_terminated_length": 176.5, "completions/min_length": 169.0, "completions/min_terminated_length": 169.0, "entropy": 0.27090173214673996, "epoch": 0.09061332690685625, "frac_reward_zero_std": 0.0, "grad_norm": 2.849970817565918, "learning_rate": 3.1666666666666667e-06, "loss": 0.0115, "num_tokens": 5100800.0, "reward": 0.7512168884277344, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998549222946167, "reward_meter_std": 0.00034224658156745136, "reward_repeat_penalty_mean": 0.9027777910232544, "reward_repeat_penalty_std": 0.07120776921510696, "reward_std": 0.05916035920381546, "reward_total_composite_mean": 0.7512168884277344, "reward_total_composite_std": 0.05916035175323486, "reward_total_mean": 0.7512168884277344, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998549222946167, "rewards/meter/std": 0.00034224658156745136, "rewards/repeat_penalty/mean": 0.9027777910232544, "rewards/repeat_penalty/std": 0.07120776921510696, "rewards/total_composite/mean": 0.7512168884277344, "rewards/total_composite/std": 0.05916035175323486, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005678415298462, "sampling/importance_sampling_ratio/min": 0.1139238253235817, "sampling/sampling_logp_difference/max": 2.1722252368927, "sampling/sampling_logp_difference/mean": 0.04010716825723648, "step": 2256 }, { "clip_ratio/high_max": 0.01619163854047656, "clip_ratio/high_mean": 0.01619163854047656, "clip_ratio/low_mean": 0.01093733258312568, "clip_ratio/low_min": 0.01093733258312568, "clip_ratio/region_mean": 0.02712897112360224, "completions/clipped_ratio": 0.0, "completions/max_length": 244.0, "completions/max_terminated_length": 244.0, "completions/mean_length": 222.625, "completions/mean_terminated_length": 222.625, "completions/min_length": 192.0, "completions/min_terminated_length": 192.0, "entropy": 0.273501418530941, "epoch": 0.0906534923886412, "frac_reward_zero_std": 0.0, "grad_norm": 2.3163866996765137, "learning_rate": 3.1636363636363635e-06, "loss": -0.0557, "num_tokens": 5104173.0, "reward": 0.7136683464050293, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8035714030265808, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.9975998401641846, "reward_meter_std": 0.0006318566738627851, "reward_repeat_penalty_mean": 0.8901515007019043, "reward_repeat_penalty_std": 0.057505469769239426, "reward_std": 0.08285731077194214, "reward_total_composite_mean": 0.7136683464050293, "reward_total_composite_std": 0.08285731077194214, "reward_total_mean": 0.7136683464050293, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8035714030265808, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.9975998401641846, "rewards/meter/std": 0.0006318566738627851, "rewards/repeat_penalty/mean": 0.8901515007019043, "rewards/repeat_penalty/std": 0.057505469769239426, "rewards/total_composite/mean": 0.7136683464050293, "rewards/total_composite/std": 0.08285731077194214, "sampling/importance_sampling_ratio/max": 1.9951218366622925, "sampling/importance_sampling_ratio/mean": 1.0057517290115356, "sampling/importance_sampling_ratio/min": 0.14766825735569, "sampling/sampling_logp_difference/max": 1.9127869606018066, "sampling/sampling_logp_difference/mean": 0.03854899853467941, "step": 2257 }, { "clip_ratio/high_max": 0.029055888997390866, "clip_ratio/high_mean": 0.029055888997390866, "clip_ratio/low_mean": 0.00781669607385993, "clip_ratio/low_min": 0.00781669607385993, "clip_ratio/region_mean": 0.036872585071250796, "completions/clipped_ratio": 0.0, "completions/max_length": 177.0, "completions/max_terminated_length": 177.0, "completions/mean_length": 173.75, "completions/mean_terminated_length": 173.75, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.29748900793492794, "epoch": 0.09069365787042616, "frac_reward_zero_std": 0.0, "grad_norm": 2.4006783962249756, "learning_rate": 3.1606060606060608e-06, "loss": 0.014, "num_tokens": 5106939.0, "reward": 0.7968448400497437, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978283047676086, "reward_meter_std": 0.0015693600289523602, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.04724043607711792, "reward_total_composite_mean": 0.7968448400497437, "reward_total_composite_std": 0.04724044352769852, "reward_total_mean": 0.7968448400497437, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978283047676086, "rewards/meter/std": 0.0015693600289523602, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.7968448400497437, "rewards/total_composite/std": 0.04724044352769852, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0058785676956177, "sampling/importance_sampling_ratio/min": 0.25441572070121765, "sampling/sampling_logp_difference/max": 1.3687856197357178, "sampling/sampling_logp_difference/mean": 0.03900068625807762, "step": 2258 }, { "clip_ratio/high_max": 0.024707219563424587, "clip_ratio/high_mean": 0.024707219563424587, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.02827864815481007, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.25, "completions/mean_terminated_length": 70.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.27262550778687, "epoch": 0.09073382335221111, "frac_reward_zero_std": 0.0, "grad_norm": 5.68310022354126, "learning_rate": 3.1575757575757576e-06, "loss": -0.0036, "num_tokens": 5108781.0, "reward": 0.9816989898681641, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9816989898681641, "reward_meter_std": 0.026165146380662918, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02616514265537262, "reward_total_composite_mean": 0.9816989898681641, "reward_total_composite_std": 0.026165146380662918, "reward_total_mean": 0.9816989898681641, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9816989898681641, "rewards/meter/std": 0.026165146380662918, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9816989898681641, "rewards/total_composite/std": 0.026165146380662918, "sampling/importance_sampling_ratio/max": 1.8370952606201172, "sampling/importance_sampling_ratio/mean": 1.0080952644348145, "sampling/importance_sampling_ratio/min": 0.35631778836250305, "sampling/sampling_logp_difference/max": 1.0319323539733887, "sampling/sampling_logp_difference/mean": 0.031211039051413536, "step": 2259 }, { "clip_ratio/high_max": 0.0209878021851182, "clip_ratio/high_mean": 0.0209878021851182, "clip_ratio/low_mean": 0.005656401859596372, "clip_ratio/low_min": 0.005656401859596372, "clip_ratio/region_mean": 0.02664420404471457, "completions/clipped_ratio": 0.0, "completions/max_length": 369.0, "completions/max_terminated_length": 369.0, "completions/mean_length": 355.625, "completions/mean_terminated_length": 355.625, "completions/min_length": 346.0, "completions/min_terminated_length": 346.0, "entropy": 0.2432896252721548, "epoch": 0.09077398883399607, "frac_reward_zero_std": 0.0, "grad_norm": 1.6112135648727417, "learning_rate": 3.1545454545454545e-06, "loss": -0.0005, "num_tokens": 5113402.0, "reward": 0.5181742906570435, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6428571343421936, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9965711236000061, "reward_meter_std": 0.0014478074153885245, "reward_repeat_penalty_mean": 0.8088235259056091, "reward_repeat_penalty_std": 0.07539646327495575, "reward_std": 0.048295676708221436, "reward_total_composite_mean": 0.5181742906570435, "reward_total_composite_std": 0.04829566553235054, "reward_total_mean": 0.5181742906570435, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6428571343421936, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9965711236000061, "rewards/meter/std": 0.0014478074153885245, "rewards/repeat_penalty/mean": 0.8088235259056091, "rewards/repeat_penalty/std": 0.07539646327495575, "rewards/total_composite/mean": 0.5181742906570435, "rewards/total_composite/std": 0.04829566553235054, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0037370920181274, "sampling/importance_sampling_ratio/min": 0.08077196031808853, "sampling/sampling_logp_difference/max": 2.516125440597534, "sampling/sampling_logp_difference/mean": 0.034783411771059036, "step": 2260 }, { "clip_ratio/high_max": 0.04133175266906619, "clip_ratio/high_mean": 0.04133175266906619, "clip_ratio/low_mean": 0.00589622650295496, "clip_ratio/low_min": 0.00589622650295496, "clip_ratio/region_mean": 0.04722797917202115, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 104.125, "completions/mean_terminated_length": 104.125, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.34618273936212063, "epoch": 0.09081415431578102, "frac_reward_zero_std": 0.0, "grad_norm": 3.7612602710723877, "learning_rate": 3.1515151515151517e-06, "loss": 0.0152, "num_tokens": 5115515.0, "reward": 0.9735202789306641, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984878301620483, "reward_meter_std": 0.000643216073513031, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07053548842668533, "reward_total_composite_mean": 0.9735202789306641, "reward_total_composite_std": 0.07053547352552414, "reward_total_mean": 0.9735202789306641, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984878301620483, "rewards/meter/std": 0.000643216073513031, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9735202789306641, "rewards/total_composite/std": 0.07053547352552414, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0051672458648682, "sampling/importance_sampling_ratio/min": 0.07617860287427902, "sampling/sampling_logp_difference/max": 2.574674606323242, "sampling/sampling_logp_difference/mean": 0.04828030988574028, "step": 2261 }, { "clip_ratio/high_max": 0.0202221788931638, "clip_ratio/high_mean": 0.0202221788931638, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/region_mean": 0.022033773129805923, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.375, "completions/mean_terminated_length": 68.375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.19510209374129772, "epoch": 0.09085431979756597, "frac_reward_zero_std": 0.0, "grad_norm": 3.4450576305389404, "learning_rate": 3.1484848484848485e-06, "loss": 0.0082, "num_tokens": 5117374.0, "reward": 0.9669647812843323, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9669647812843323, "reward_meter_std": 0.014694334007799625, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014694339595735073, "reward_total_composite_mean": 0.9669647812843323, "reward_total_composite_std": 0.014694334007799625, "reward_total_mean": 0.9669647812843323, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9669647812843323, "rewards/meter/std": 0.014694334007799625, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9669647812843323, "rewards/total_composite/std": 0.014694334007799625, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0112721920013428, "sampling/importance_sampling_ratio/min": 0.3059942126274109, "sampling/sampling_logp_difference/max": 1.1841890811920166, "sampling/sampling_logp_difference/mean": 0.028783582150936127, "step": 2262 }, { "clip_ratio/high_max": 0.00390625, "clip_ratio/high_mean": 0.00390625, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.007694128900766373, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.125, "completions/mean_terminated_length": 32.125, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.05214253021404147, "epoch": 0.09089448527935093, "frac_reward_zero_std": 0.0, "grad_norm": 1.744556188583374, "learning_rate": 3.145454545454546e-06, "loss": 0.0034, "num_tokens": 5118823.0, "reward": 0.9992284774780273, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992284774780273, "reward_meter_std": 0.00014189988723956048, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00014191023365128785, "reward_total_composite_mean": 0.9992284774780273, "reward_total_composite_std": 0.00014189988723956048, "reward_total_mean": 0.9992284774780273, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992284774780273, "rewards/meter/std": 0.00014189988723956048, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992284774780273, "rewards/total_composite/std": 0.00014189988723956048, "sampling/importance_sampling_ratio/max": 1.1715004444122314, "sampling/importance_sampling_ratio/mean": 0.9978591203689575, "sampling/importance_sampling_ratio/min": 0.8397412300109863, "sampling/sampling_logp_difference/max": 0.17466145753860474, "sampling/sampling_logp_difference/mean": 0.005864746868610382, "step": 2263 }, { "clip_ratio/high_max": 0.014823656994849443, "clip_ratio/high_mean": 0.014823656994849443, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.018611535895615816, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 33.375, "completions/mean_terminated_length": 33.375, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.16098480112850666, "epoch": 0.09093465076113588, "frac_reward_zero_std": 0.0, "grad_norm": 4.9907097816467285, "learning_rate": 3.142424242424243e-06, "loss": -0.0007, "num_tokens": 5120274.0, "reward": 0.9775052070617676, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9775052070617676, "reward_meter_std": 0.010550287552177906, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010550293140113354, "reward_total_composite_mean": 0.9775052070617676, "reward_total_composite_std": 0.010550287552177906, "reward_total_mean": 0.9775052070617676, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9775052070617676, "rewards/meter/std": 0.010550287552177906, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9775052070617676, "rewards/total_composite/std": 0.010550287552177906, "sampling/importance_sampling_ratio/max": 1.8347361087799072, "sampling/importance_sampling_ratio/mean": 1.0072755813598633, "sampling/importance_sampling_ratio/min": 0.528147280216217, "sampling/sampling_logp_difference/max": 0.6383800506591797, "sampling/sampling_logp_difference/mean": 0.02235337160527706, "step": 2264 }, { "clip_ratio/high_max": 0.00745840510353446, "clip_ratio/high_mean": 0.00745840510353446, "clip_ratio/low_mean": 0.01339285762514919, "clip_ratio/low_min": 0.01339285762514919, "clip_ratio/region_mean": 0.02085126272868365, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 83.625, "completions/mean_terminated_length": 83.625, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.08839869406074286, "epoch": 0.09097481624292084, "frac_reward_zero_std": 0.0, "grad_norm": 2.8727030754089355, "learning_rate": 3.13939393939394e-06, "loss": 0.0095, "num_tokens": 5122367.0, "reward": 0.9946054816246033, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946054816246033, "reward_meter_std": 0.0010328061180189252, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010328061180189252, "reward_total_composite_mean": 0.9946054816246033, "reward_total_composite_std": 0.0010328061180189252, "reward_total_mean": 0.9946054816246033, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946054816246033, "rewards/meter/std": 0.0010328061180189252, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946054816246033, "rewards/total_composite/std": 0.0010328061180189252, "sampling/importance_sampling_ratio/max": 1.7508782148361206, "sampling/importance_sampling_ratio/mean": 1.0014218091964722, "sampling/importance_sampling_ratio/min": 0.23498652875423431, "sampling/sampling_logp_difference/max": 1.4482271671295166, "sampling/sampling_logp_difference/mean": 0.018597107380628586, "step": 2265 }, { "clip_ratio/high_max": 0.007080778945237398, "clip_ratio/high_mean": 0.007080778945237398, "clip_ratio/low_mean": 0.006256514578126371, "clip_ratio/low_min": 0.006256514578126371, "clip_ratio/region_mean": 0.013337293523363769, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 140.125, "completions/mean_terminated_length": 140.125, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.09540552459657192, "epoch": 0.09101498172470579, "frac_reward_zero_std": 0.0, "grad_norm": 3.1527485847473145, "learning_rate": 3.1363636363636367e-06, "loss": 0.0001, "num_tokens": 5125064.0, "reward": 0.81532883644104, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949551820755005, "reward_meter_std": 0.0007479682681150734, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.057526424527168274, "reward_total_composite_mean": 0.81532883644104, "reward_total_composite_std": 0.05752642825245857, "reward_total_mean": 0.81532883644104, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949551820755005, "rewards/meter/std": 0.0007479682681150734, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.81532883644104, "rewards/total_composite/std": 0.05752642825245857, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0007151365280151, "sampling/importance_sampling_ratio/min": 0.32793816924095154, "sampling/sampling_logp_difference/max": 1.1149301528930664, "sampling/sampling_logp_difference/mean": 0.01804959587752819, "step": 2266 }, { "clip_ratio/high_max": 0.012057766725774854, "clip_ratio/high_mean": 0.012057766725774854, "clip_ratio/low_mean": 0.006448142405133694, "clip_ratio/low_min": 0.006448142405133694, "clip_ratio/region_mean": 0.01850590913090855, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 135.125, "completions/mean_terminated_length": 135.125, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.09932332392781973, "epoch": 0.09105514720649074, "frac_reward_zero_std": 0.0, "grad_norm": 2.1978232860565186, "learning_rate": 3.133333333333334e-06, "loss": 0.0082, "num_tokens": 5127489.0, "reward": 0.9982975721359253, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982975721359253, "reward_meter_std": 0.000450806604931131, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00045079842675477266, "reward_total_composite_mean": 0.9982975721359253, "reward_total_composite_std": 0.000450806604931131, "reward_total_mean": 0.9982975721359253, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982975721359253, "rewards/meter/std": 0.000450806604931131, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982975721359253, "rewards/total_composite/std": 0.000450806604931131, "sampling/importance_sampling_ratio/max": 1.8831219673156738, "sampling/importance_sampling_ratio/mean": 0.9998656511306763, "sampling/importance_sampling_ratio/min": 0.2570839524269104, "sampling/sampling_logp_difference/max": 1.3583526611328125, "sampling/sampling_logp_difference/mean": 0.022603249177336693, "step": 2267 }, { "clip_ratio/high_max": 0.005474452394992113, "clip_ratio/high_mean": 0.005474452394992113, "clip_ratio/low_mean": 0.013712125248275697, "clip_ratio/low_min": 0.013712125248275697, "clip_ratio/region_mean": 0.01918657764326781, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 135.75, "completions/mean_terminated_length": 135.75, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.1438667206093669, "epoch": 0.0910953126882757, "frac_reward_zero_std": 0.0, "grad_norm": 3.012678384780884, "learning_rate": 3.130303030303031e-06, "loss": 0.0066, "num_tokens": 5130007.0, "reward": 0.8470007181167603, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9679410457611084, "reward_meter_std": 0.006429874338209629, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05034322291612625, "reward_total_composite_mean": 0.8470007181167603, "reward_total_composite_std": 0.05034321919083595, "reward_total_mean": 0.8470007181167603, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9679410457611084, "rewards/meter/std": 0.006429874338209629, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8470007181167603, "rewards/total_composite/std": 0.05034321919083595, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0043870210647583, "sampling/importance_sampling_ratio/min": 0.3170107305049896, "sampling/sampling_logp_difference/max": 1.6587629318237305, "sampling/sampling_logp_difference/mean": 0.022967170923948288, "step": 2268 }, { "clip_ratio/high_max": 0.012555458233691752, "clip_ratio/high_mean": 0.012555458233691752, "clip_ratio/low_mean": 0.0028210257878527045, "clip_ratio/low_min": 0.0028210257878527045, "clip_ratio/region_mean": 0.015376484021544456, "completions/clipped_ratio": 0.0, "completions/max_length": 181.0, "completions/max_terminated_length": 181.0, "completions/mean_length": 178.75, "completions/mean_terminated_length": 178.75, "completions/min_length": 176.0, "completions/min_terminated_length": 176.0, "entropy": 0.19854111224412918, "epoch": 0.09113547817006065, "frac_reward_zero_std": 0.0, "grad_norm": 2.066251754760742, "learning_rate": 3.1272727272727276e-06, "loss": 0.0004, "num_tokens": 5133013.0, "reward": 0.7400156855583191, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990208148956299, "reward_meter_std": 0.00016956882609520108, "reward_repeat_penalty_mean": 0.8888888955116272, "reward_repeat_penalty_std": 0.10286889225244522, "reward_std": 0.08564040064811707, "reward_total_composite_mean": 0.7400156855583191, "reward_total_composite_std": 0.08564041554927826, "reward_total_mean": 0.7400156855583191, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990208148956299, "rewards/meter/std": 0.00016956882609520108, "rewards/repeat_penalty/mean": 0.8888888955116272, "rewards/repeat_penalty/std": 0.10286889225244522, "rewards/total_composite/mean": 0.7400156855583191, "rewards/total_composite/std": 0.08564041554927826, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0039939880371094, "sampling/importance_sampling_ratio/min": 0.2306935042142868, "sampling/sampling_logp_difference/max": 1.466665267944336, "sampling/sampling_logp_difference/mean": 0.023759307339787483, "step": 2269 }, { "clip_ratio/high_max": 0.02643707417882979, "clip_ratio/high_mean": 0.02643707417882979, "clip_ratio/low_mean": 0.01330830343067646, "clip_ratio/low_min": 0.01330830343067646, "clip_ratio/region_mean": 0.03974537760950625, "completions/clipped_ratio": 0.0, "completions/max_length": 177.0, "completions/max_terminated_length": 177.0, "completions/mean_length": 170.25, "completions/mean_terminated_length": 170.25, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.38951336964964867, "epoch": 0.0911756436518456, "frac_reward_zero_std": 0.0, "grad_norm": 3.698033094406128, "learning_rate": 3.1242424242424245e-06, "loss": 0.0089, "num_tokens": 5136127.0, "reward": 0.7962738275527954, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971339702606201, "reward_meter_std": 0.0020925395656377077, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.04691069573163986, "reward_total_composite_mean": 0.7962738275527954, "reward_total_composite_std": 0.04691069945693016, "reward_total_mean": 0.7962738275527954, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971339702606201, "rewards/meter/std": 0.0020925395656377077, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.7962738275527954, "rewards/total_composite/std": 0.04691069945693016, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.011202096939087, "sampling/importance_sampling_ratio/min": 0.05918107181787491, "sampling/sampling_logp_difference/max": 2.827153444290161, "sampling/sampling_logp_difference/mean": 0.0504683218896389, "step": 2270 }, { "clip_ratio/high_max": 0.026904894039034843, "clip_ratio/high_mean": 0.026904894039034843, "clip_ratio/low_mean": 0.0624148678034544, "clip_ratio/low_min": 0.0624148678034544, "clip_ratio/region_mean": 0.08931976184248924, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 62.125, "completions/mean_terminated_length": 62.125, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.3145434930920601, "epoch": 0.09121580913363056, "frac_reward_zero_std": 0.0, "grad_norm": 6.407037258148193, "learning_rate": 3.1212121212121217e-06, "loss": 0.0327, "num_tokens": 5137976.0, "reward": 0.09545363485813141, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.09545363485813141, "reward_meter_std": 0.17621159553527832, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.17621161043643951, "reward_total_composite_mean": 0.09545363485813141, "reward_total_composite_std": 0.17621159553527832, "reward_total_mean": 0.09545363485813141, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.09545363485813141, "rewards/meter/std": 0.17621159553527832, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.09545363485813141, "rewards/total_composite/std": 0.17621159553527832, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9893724918365479, "sampling/importance_sampling_ratio/min": 0.05866949260234833, "sampling/sampling_logp_difference/max": 2.8358354568481445, "sampling/sampling_logp_difference/mean": 0.09075994789600372, "step": 2271 }, { "clip_ratio/high_max": 0.01436411531176418, "clip_ratio/high_mean": 0.01436411531176418, "clip_ratio/low_mean": 0.00911509501747787, "clip_ratio/low_min": 0.00911509501747787, "clip_ratio/region_mean": 0.02347921032924205, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.0, "completions/mean_terminated_length": 69.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.21663649939000607, "epoch": 0.09125597461541551, "frac_reward_zero_std": 0.0, "grad_norm": 5.84548282623291, "learning_rate": 3.1181818181818186e-06, "loss": -0.0059, "num_tokens": 5139832.0, "reward": 0.9690738916397095, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9690738916397095, "reward_meter_std": 0.02259281650185585, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.022592833265662193, "reward_total_composite_mean": 0.9690738916397095, "reward_total_composite_std": 0.02259281650185585, "reward_total_mean": 0.9690738916397095, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9690738916397095, "rewards/meter/std": 0.02259281650185585, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9690738916397095, "rewards/total_composite/std": 0.02259281650185585, "sampling/importance_sampling_ratio/max": 1.7162621021270752, "sampling/importance_sampling_ratio/mean": 1.0030878782272339, "sampling/importance_sampling_ratio/min": 0.33046165108680725, "sampling/sampling_logp_difference/max": 1.107264757156372, "sampling/sampling_logp_difference/mean": 0.031493693590164185, "step": 2272 }, { "clip_ratio/high_max": 0.008914577425457537, "clip_ratio/high_mean": 0.008914577425457537, "clip_ratio/low_mean": 0.011750375153496861, "clip_ratio/low_min": 0.011750375153496861, "clip_ratio/region_mean": 0.0206649525789544, "completions/clipped_ratio": 0.0, "completions/max_length": 318.0, "completions/max_terminated_length": 318.0, "completions/mean_length": 308.125, "completions/mean_terminated_length": 308.125, "completions/min_length": 295.0, "completions/min_terminated_length": 295.0, "entropy": 0.16073014307767153, "epoch": 0.09129614009720047, "frac_reward_zero_std": 0.0, "grad_norm": 2.4946935176849365, "learning_rate": 3.1151515151515154e-06, "loss": 0.0065, "num_tokens": 5144049.0, "reward": 0.5205361843109131, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6634615659713745, "reward_count_adherence_std": 0.039811473339796066, "reward_meter_mean": 0.985474705696106, "reward_meter_std": 0.02723194845020771, "reward_repeat_penalty_mean": 0.7977941036224365, "reward_repeat_penalty_std": 0.08265344798564911, "reward_std": 0.05346594378352165, "reward_total_composite_mean": 0.5205361843109131, "reward_total_composite_std": 0.053465962409973145, "reward_total_mean": 0.5205361843109131, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6634615659713745, "rewards/count_adherence/std": 0.039811473339796066, "rewards/meter/mean": 0.985474705696106, "rewards/meter/std": 0.02723194845020771, "rewards/repeat_penalty/mean": 0.7977941036224365, "rewards/repeat_penalty/std": 0.08265344798564911, "rewards/total_composite/mean": 0.5205361843109131, "rewards/total_composite/std": 0.053465962409973145, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0016851425170898, "sampling/importance_sampling_ratio/min": 0.26280176639556885, "sampling/sampling_logp_difference/max": 1.336355209350586, "sampling/sampling_logp_difference/mean": 0.02342665195465088, "step": 2273 }, { "clip_ratio/high_max": 0.02496329549467191, "clip_ratio/high_mean": 0.02496329549467191, "clip_ratio/low_mean": 0.0016666667070239782, "clip_ratio/low_min": 0.0016666667070239782, "clip_ratio/region_mean": 0.02662996220169589, "completions/clipped_ratio": 0.0, "completions/max_length": 158.0, "completions/max_terminated_length": 158.0, "completions/mean_length": 154.0, "completions/mean_terminated_length": 154.0, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "entropy": 0.29750954918563366, "epoch": 0.09133630557898542, "frac_reward_zero_std": 0.0, "grad_norm": 2.2683939933776855, "learning_rate": 3.1121212121212126e-06, "loss": -0.006, "num_tokens": 5146825.0, "reward": 0.976262092590332, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939357042312622, "reward_meter_std": 0.003742816159501672, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05179871618747711, "reward_total_composite_mean": 0.976262092590332, "reward_total_composite_std": 0.051798708736896515, "reward_total_mean": 0.976262092590332, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939357042312622, "rewards/meter/std": 0.003742816159501672, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.976262092590332, "rewards/total_composite/std": 0.051798708736896515, "sampling/importance_sampling_ratio/max": 1.880401372909546, "sampling/importance_sampling_ratio/mean": 1.006531834602356, "sampling/importance_sampling_ratio/min": 0.27912476658821106, "sampling/sampling_logp_difference/max": 1.2760963439941406, "sampling/sampling_logp_difference/mean": 0.03507037088274956, "step": 2274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.012424270971678197, "epoch": 0.09137647106077038, "frac_reward_zero_std": 0.0, "grad_norm": 2.3654937744140625, "learning_rate": 3.1090909090909095e-06, "loss": 0.0025, "num_tokens": 5148537.0, "reward": 0.7876189947128296, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7876189947128296, "reward_meter_std": 3.669682701001875e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.669681245810352e-05, "reward_total_composite_mean": 0.7876189947128296, "reward_total_composite_std": 3.669682701001875e-05, "reward_total_mean": 0.7876189947128296, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7876189947128296, "rewards/meter/std": 3.669682701001875e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876189947128296, "rewards/total_composite/std": 3.669682701001875e-05, "sampling/importance_sampling_ratio/max": 1.2923461198806763, "sampling/importance_sampling_ratio/mean": 1.0020716190338135, "sampling/importance_sampling_ratio/min": 0.9631562829017639, "sampling/sampling_logp_difference/max": 0.25645923614501953, "sampling/sampling_logp_difference/mean": 0.0021758992224931717, "step": 2275 }, { "clip_ratio/high_max": 0.009610342094674706, "clip_ratio/high_mean": 0.009610342094674706, "clip_ratio/low_mean": 0.0008865247946232557, "clip_ratio/low_min": 0.0008865247946232557, "clip_ratio/region_mean": 0.010496866889297962, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 142.375, "completions/mean_terminated_length": 142.375, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 0.07997249066829681, "epoch": 0.09141663654255533, "frac_reward_zero_std": 0.0, "grad_norm": 1.912965178489685, "learning_rate": 3.1060606060606063e-06, "loss": -0.0014, "num_tokens": 5151308.0, "reward": 0.8573694229125977, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956492185592651, "reward_meter_std": 0.00017072130867745727, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.05143444612622261, "reward_std": 0.05129777267575264, "reward_total_composite_mean": 0.8573694229125977, "reward_total_composite_std": 0.05129777640104294, "reward_total_mean": 0.8573694229125977, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956492185592651, "rewards/meter/std": 0.00017072130867745727, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.05143444612622261, "rewards/total_composite/mean": 0.8573694229125977, "rewards/total_composite/std": 0.05129777640104294, "sampling/importance_sampling_ratio/max": 1.4935818910598755, "sampling/importance_sampling_ratio/mean": 1.000401258468628, "sampling/importance_sampling_ratio/min": 0.3668079972267151, "sampling/sampling_logp_difference/max": 1.0029168128967285, "sampling/sampling_logp_difference/mean": 0.01284986175596714, "step": 2276 }, { "clip_ratio/high_max": 0.01534117921255529, "clip_ratio/high_mean": 0.01534117921255529, "clip_ratio/low_mean": 0.009245004854165018, "clip_ratio/low_min": 0.009245004854165018, "clip_ratio/region_mean": 0.024586184066720307, "completions/clipped_ratio": 0.0, "completions/max_length": 205.0, "completions/max_terminated_length": 205.0, "completions/mean_length": 170.0, "completions/mean_terminated_length": 170.0, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "entropy": 0.18850434944033623, "epoch": 0.09145680202434028, "frac_reward_zero_std": 0.0, "grad_norm": 2.30692195892334, "learning_rate": 3.103030303030303e-06, "loss": -0.0004, "num_tokens": 5154060.0, "reward": 0.658399760723114, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.9647563695907593, "reward_meter_std": 0.03250608593225479, "reward_repeat_penalty_mean": 0.7992424368858337, "reward_repeat_penalty_std": 0.11809822916984558, "reward_std": 0.10351462662220001, "reward_total_composite_mean": 0.658399760723114, "reward_total_composite_std": 0.10351461172103882, "reward_total_mean": 0.658399760723114, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.9647563695907593, "rewards/meter/std": 0.03250608593225479, "rewards/repeat_penalty/mean": 0.7992424368858337, "rewards/repeat_penalty/std": 0.11809822916984558, "rewards/total_composite/mean": 0.658399760723114, "rewards/total_composite/std": 0.10351461172103882, "sampling/importance_sampling_ratio/max": 1.7825084924697876, "sampling/importance_sampling_ratio/mean": 0.9995459318161011, "sampling/importance_sampling_ratio/min": 0.20509281754493713, "sampling/sampling_logp_difference/max": 1.5842926502227783, "sampling/sampling_logp_difference/mean": 0.02872909978032112, "step": 2277 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.006176002731081098, "epoch": 0.09149696750612524, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.1000000000000004e-06, "loss": 0.0, "num_tokens": 5155988.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.023223876953125, "sampling/importance_sampling_ratio/mean": 1.0005130767822266, "sampling/importance_sampling_ratio/min": 0.9780369400978088, "sampling/sampling_logp_difference/max": 0.022958219051361084, "sampling/sampling_logp_difference/mean": 0.000632537470664829, "step": 2278 }, { "clip_ratio/high_max": 0.04533730214461684, "clip_ratio/high_mean": 0.04533730214461684, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/region_mean": 0.048809524392709136, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 35.875, "completions/mean_terminated_length": 35.875, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.3067398630082607, "epoch": 0.09153713298791019, "frac_reward_zero_std": 0.0, "grad_norm": 13.398720741271973, "learning_rate": 3.0969696969696972e-06, "loss": 0.0082, "num_tokens": 5157395.0, "reward": 0.9567666053771973, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9567666053771973, "reward_meter_std": 0.09672017395496368, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09672017395496368, "reward_total_composite_mean": 0.9567666053771973, "reward_total_composite_std": 0.09672017395496368, "reward_total_mean": 0.9567666053771973, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9567666053771973, "rewards/meter/std": 0.09672017395496368, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9567666053771973, "rewards/total_composite/std": 0.09672017395496368, "sampling/importance_sampling_ratio/max": 1.522891640663147, "sampling/importance_sampling_ratio/mean": 0.9919366836547852, "sampling/importance_sampling_ratio/min": 0.2881072461605072, "sampling/sampling_logp_difference/max": 1.244422435760498, "sampling/sampling_logp_difference/mean": 0.055583853274583817, "step": 2279 }, { "clip_ratio/high_max": 0.008134586038067937, "clip_ratio/high_mean": 0.008134586038067937, "clip_ratio/low_mean": 0.009806309011764824, "clip_ratio/low_min": 0.009806309011764824, "clip_ratio/region_mean": 0.01794089504983276, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 137.625, "completions/mean_terminated_length": 137.625, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.18117596302181482, "epoch": 0.09157729846969515, "frac_reward_zero_std": 0.0, "grad_norm": 5.9251813888549805, "learning_rate": 3.093939393939394e-06, "loss": 0.0149, "num_tokens": 5159888.0, "reward": 0.8324918150901794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9712404012680054, "reward_meter_std": 0.020027272403240204, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.017166240140795708, "reward_total_composite_mean": 0.8324918150901794, "reward_total_composite_std": 0.017166240140795708, "reward_total_mean": 0.8324918150901794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9712404012680054, "rewards/meter/std": 0.020027272403240204, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8324918150901794, "rewards/total_composite/std": 0.017166240140795708, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007045865058899, "sampling/importance_sampling_ratio/min": 0.2323206663131714, "sampling/sampling_logp_difference/max": 1.4596366882324219, "sampling/sampling_logp_difference/mean": 0.03175117075443268, "step": 2280 }, { "clip_ratio/high_max": 0.015625, "clip_ratio/high_mean": 0.015625, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/region_mean": 0.023200757801532745, "completions/clipped_ratio": 0.0, "completions/max_length": 33.0, "completions/max_terminated_length": 33.0, "completions/mean_length": 32.25, "completions/mean_terminated_length": 32.25, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.057134561240673065, "epoch": 0.0916174639514801, "frac_reward_zero_std": 0.0, "grad_norm": 2.3620288372039795, "learning_rate": 3.090909090909091e-06, "loss": 0.0073, "num_tokens": 5161346.0, "reward": 0.9992298483848572, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992298483848572, "reward_meter_std": 7.94332881923765e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.940443902043626e-05, "reward_total_composite_mean": 0.9992298483848572, "reward_total_composite_std": 7.94332881923765e-05, "reward_total_mean": 0.9992298483848572, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992298483848572, "rewards/meter/std": 7.94332881923765e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992298483848572, "rewards/total_composite/std": 7.94332881923765e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0008995532989502, "sampling/importance_sampling_ratio/min": 0.5924263596534729, "sampling/sampling_logp_difference/max": 0.7187650203704834, "sampling/sampling_logp_difference/mean": 0.016870969906449318, "step": 2281 }, { "clip_ratio/high_max": 0.01910717412829399, "clip_ratio/high_mean": 0.01910717412829399, "clip_ratio/low_mean": 0.005548747314605862, "clip_ratio/low_min": 0.005548747314605862, "clip_ratio/region_mean": 0.024655921442899853, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 132.25, "completions/mean_terminated_length": 132.25, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.23136323876678944, "epoch": 0.09165762943326505, "frac_reward_zero_std": 0.0, "grad_norm": 2.3963494300842285, "learning_rate": 3.087878787878788e-06, "loss": 0.0173, "num_tokens": 5163820.0, "reward": 0.8361626863479614, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9742553234100342, "reward_meter_std": 0.023107483983039856, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.1322600096464157, "reward_std": 0.13729579746723175, "reward_total_composite_mean": 0.8361626863479614, "reward_total_composite_std": 0.13729579746723175, "reward_total_mean": 0.8361626863479614, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9742553234100342, "rewards/meter/std": 0.023107483983039856, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.1322600096464157, "rewards/total_composite/mean": 0.8361626863479614, "rewards/total_composite/std": 0.13729579746723175, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0067442655563354, "sampling/importance_sampling_ratio/min": 0.22720544040203094, "sampling/sampling_logp_difference/max": 1.481900691986084, "sampling/sampling_logp_difference/mean": 0.03228021413087845, "step": 2282 }, { "clip_ratio/high_max": 0.012570029124617577, "clip_ratio/high_mean": 0.012570029124617577, "clip_ratio/low_mean": 0.008858592598699033, "clip_ratio/low_min": 0.008858592598699033, "clip_ratio/region_mean": 0.02142862172331661, "completions/clipped_ratio": 0.0, "completions/max_length": 171.0, "completions/max_terminated_length": 171.0, "completions/mean_length": 168.875, "completions/mean_terminated_length": 168.875, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "entropy": 0.11252321023494005, "epoch": 0.09169779491505001, "frac_reward_zero_std": 0.0, "grad_norm": 2.038731098175049, "learning_rate": 3.084848484848485e-06, "loss": 0.0041, "num_tokens": 5166635.0, "reward": 0.8599779009819031, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957842230796814, "reward_meter_std": 0.0002837553038261831, "reward_repeat_penalty_mean": 0.8636363744735718, "reward_repeat_penalty_std": 0.08416546881198883, "reward_std": 0.08360189944505692, "reward_total_composite_mean": 0.8599779009819031, "reward_total_composite_std": 0.08360190689563751, "reward_total_mean": 0.8599779009819031, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957842230796814, "rewards/meter/std": 0.0002837553038261831, "rewards/repeat_penalty/mean": 0.8636363744735718, "rewards/repeat_penalty/std": 0.08416546881198883, "rewards/total_composite/mean": 0.8599779009819031, "rewards/total_composite/std": 0.08360190689563751, "sampling/importance_sampling_ratio/max": 1.8878965377807617, "sampling/importance_sampling_ratio/mean": 1.0022042989730835, "sampling/importance_sampling_ratio/min": 0.22793714702129364, "sampling/sampling_logp_difference/max": 1.4786853790283203, "sampling/sampling_logp_difference/mean": 0.021027982234954834, "step": 2283 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.0025252392806578428, "epoch": 0.09173796039683496, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.081818181818182e-06, "loss": 0.0, "num_tokens": 5168059.0, "reward": 0.9929623007774353, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9929623007774353, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9929623007774353, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9929623007774353, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9929623007774353, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9929623007774353, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0076669454574585, "sampling/importance_sampling_ratio/mean": 1.0002309083938599, "sampling/importance_sampling_ratio/min": 0.9986435174942017, "sampling/sampling_logp_difference/max": 0.0076376767829060555, "sampling/sampling_logp_difference/mean": 0.0002442288678139448, "step": 2284 }, { "clip_ratio/high_max": 0.006868339143693447, "clip_ratio/high_mean": 0.006868339143693447, "clip_ratio/low_mean": 0.014906731550581753, "clip_ratio/low_min": 0.014906731550581753, "clip_ratio/region_mean": 0.0217750706942752, "completions/clipped_ratio": 0.0, "completions/max_length": 183.0, "completions/max_terminated_length": 183.0, "completions/mean_length": 171.75, "completions/mean_terminated_length": 171.75, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.19338966719806194, "epoch": 0.09177812587861992, "frac_reward_zero_std": 0.0, "grad_norm": 1.7548184394836426, "learning_rate": 3.078787878787879e-06, "loss": -0.0656, "num_tokens": 5170897.0, "reward": 0.8935445547103882, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.09258200973272324, "reward_meter_mean": 0.9989954233169556, "reward_meter_std": 0.0003574988222680986, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.07573915272951126, "reward_total_composite_mean": 0.8935445547103882, "reward_total_composite_std": 0.07573916018009186, "reward_total_mean": 0.8935445547103882, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.09258200973272324, "rewards/meter/mean": 0.9989954233169556, "rewards/meter/std": 0.0003574988222680986, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.8935445547103882, "rewards/total_composite/std": 0.07573916018009186, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0065274238586426, "sampling/importance_sampling_ratio/min": 0.13853120803833008, "sampling/sampling_logp_difference/max": 1.9766596555709839, "sampling/sampling_logp_difference/mean": 0.022320233285427094, "step": 2285 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/region_mean": 0.002358490601181984, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 106.0, "completions/mean_terminated_length": 106.0, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.009184476220980287, "epoch": 0.09181829136040487, "frac_reward_zero_std": 0.0, "grad_norm": 1.9542547464370728, "learning_rate": 3.075757575757576e-06, "loss": -0.0008, "num_tokens": 5173297.0, "reward": 0.5830118656158447, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.680180549621582, "reward_meter_std": 0.013479826971888542, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011554128490388393, "reward_total_composite_mean": 0.5830118656158447, "reward_total_composite_std": 0.011554129421710968, "reward_total_mean": 0.5830118656158447, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.680180549621582, "rewards/meter/std": 0.013479826971888542, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5830118656158447, "rewards/total_composite/std": 0.011554129421710968, "sampling/importance_sampling_ratio/max": 1.85325026512146, "sampling/importance_sampling_ratio/mean": 1.0005828142166138, "sampling/importance_sampling_ratio/min": 0.48308974504470825, "sampling/sampling_logp_difference/max": 0.7275528907775879, "sampling/sampling_logp_difference/mean": 0.0026654843240976334, "step": 2286 }, { "clip_ratio/high_max": 0.016635622712783515, "clip_ratio/high_mean": 0.016635622712783515, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/region_mean": 0.018421337008476257, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 73.875, "completions/mean_terminated_length": 73.875, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.24043426476418972, "epoch": 0.09185845684218982, "frac_reward_zero_std": 0.0, "grad_norm": 5.288448810577393, "learning_rate": 3.0727272727272727e-06, "loss": -0.0098, "num_tokens": 5175104.0, "reward": 0.8793976306915283, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8793976306915283, "reward_meter_std": 0.33859992027282715, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.33859992027282715, "reward_total_composite_mean": 0.8793976306915283, "reward_total_composite_std": 0.33859992027282715, "reward_total_mean": 0.8793976306915283, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8793976306915283, "rewards/meter/std": 0.33859992027282715, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8793976306915283, "rewards/total_composite/std": 0.33859992027282715, "sampling/importance_sampling_ratio/max": 1.7006129026412964, "sampling/importance_sampling_ratio/mean": 1.005669355392456, "sampling/importance_sampling_ratio/min": 0.39154288172721863, "sampling/sampling_logp_difference/max": 0.9376602172851562, "sampling/sampling_logp_difference/mean": 0.027358917519450188, "step": 2287 }, { "clip_ratio/high_max": 0.017823605565354228, "clip_ratio/high_mean": 0.017823605565354228, "clip_ratio/low_mean": 0.008171421475708485, "clip_ratio/low_min": 0.008171421475708485, "clip_ratio/region_mean": 0.025995027041062713, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 168.0, "completions/mean_terminated_length": 168.0, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.17776653543114662, "epoch": 0.09189862232397478, "frac_reward_zero_std": 0.0, "grad_norm": 7.310388088226318, "learning_rate": 3.0696969696969696e-06, "loss": 0.013, "num_tokens": 5178040.0, "reward": 0.621074378490448, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7916620969772339, "reward_meter_std": 0.292292982339859, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.2475537210702896, "reward_total_composite_mean": 0.621074378490448, "reward_total_composite_std": 0.2475537359714508, "reward_total_mean": 0.621074378490448, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7916620969772339, "rewards/meter/std": 0.292292982339859, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.621074378490448, "rewards/total_composite/std": 0.2475537359714508, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000993251800537, "sampling/importance_sampling_ratio/min": 0.23216624557971954, "sampling/sampling_logp_difference/max": 1.4603016376495361, "sampling/sampling_logp_difference/mean": 0.03382183238863945, "step": 2288 }, { "clip_ratio/high_max": 0.005681818351149559, "clip_ratio/high_mean": 0.005681818351149559, "clip_ratio/low_mean": 0.005434782709926367, "clip_ratio/low_min": 0.005434782709926367, "clip_ratio/region_mean": 0.011116601061075926, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.25, "completions/mean_terminated_length": 68.25, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.04147487785667181, "epoch": 0.09193878780575973, "frac_reward_zero_std": 0.0, "grad_norm": 1.6638261079788208, "learning_rate": 3.066666666666667e-06, "loss": 0.012, "num_tokens": 5179986.0, "reward": 0.9976264834403992, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976264834403992, "reward_meter_std": 0.00037816373514942825, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00037817005068063736, "reward_total_composite_mean": 0.9976264834403992, "reward_total_composite_std": 0.00037816373514942825, "reward_total_mean": 0.9976264834403992, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976264834403992, "rewards/meter/std": 0.00037816373514942825, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976264834403992, "rewards/total_composite/std": 0.00037816373514942825, "sampling/importance_sampling_ratio/max": 1.2755680084228516, "sampling/importance_sampling_ratio/mean": 0.996768593788147, "sampling/importance_sampling_ratio/min": 0.10801687091588974, "sampling/sampling_logp_difference/max": 2.2254679203033447, "sampling/sampling_logp_difference/mean": 0.01430835947394371, "step": 2289 }, { "clip_ratio/high_max": 0.005859375, "clip_ratio/high_mean": 0.005859375, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/region_mean": 0.007782451924867928, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.875, "completions/mean_terminated_length": 64.875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.04461120581254363, "epoch": 0.09197895328754468, "frac_reward_zero_std": 0.0, "grad_norm": 1.542847752571106, "learning_rate": 3.0636363636363636e-06, "loss": 0.0052, "num_tokens": 5181793.0, "reward": 0.9991232752799988, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991232752799988, "reward_meter_std": 0.00012234247697051615, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00012233950837980956, "reward_total_composite_mean": 0.9991232752799988, "reward_total_composite_std": 0.00012234247697051615, "reward_total_mean": 0.9991232752799988, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991232752799988, "rewards/meter/std": 0.00012234247697051615, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991232752799988, "rewards/total_composite/std": 0.00012234247697051615, "sampling/importance_sampling_ratio/max": 1.3870518207550049, "sampling/importance_sampling_ratio/mean": 0.9971635937690735, "sampling/importance_sampling_ratio/min": 0.27249664068222046, "sampling/sampling_logp_difference/max": 1.3001289367675781, "sampling/sampling_logp_difference/mean": 0.01243247464299202, "step": 2290 }, { "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018939394503831863, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.036194659769535065, "epoch": 0.09201911876932964, "frac_reward_zero_std": 0.0, "grad_norm": 0.26676464080810547, "learning_rate": 3.0606060606060605e-06, "loss": -0.0007, "num_tokens": 5183553.0, "reward": 0.998096764087677, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998096764087677, "reward_meter_std": 3.926327553926967e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.925973942386918e-05, "reward_total_composite_mean": 0.998096764087677, "reward_total_composite_std": 3.926327553926967e-05, "reward_total_mean": 0.998096764087677, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998096764087677, "rewards/meter/std": 3.926327553926967e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998096764087677, "rewards/total_composite/std": 3.926327553926967e-05, "sampling/importance_sampling_ratio/max": 1.0820250511169434, "sampling/importance_sampling_ratio/mean": 0.999199628829956, "sampling/importance_sampling_ratio/min": 0.4085750877857208, "sampling/sampling_logp_difference/max": 0.8950796127319336, "sampling/sampling_logp_difference/mean": 0.005647474434226751, "step": 2291 }, { "clip_ratio/high_max": 0.024386633071117103, "clip_ratio/high_mean": 0.024386633071117103, "clip_ratio/low_mean": 0.005605590064078569, "clip_ratio/low_min": 0.005605590064078569, "clip_ratio/region_mean": 0.029992223135195673, "completions/clipped_ratio": 0.0, "completions/max_length": 201.0, "completions/max_terminated_length": 201.0, "completions/mean_length": 159.25, "completions/mean_terminated_length": 159.25, "completions/min_length": 148.0, "completions/min_terminated_length": 148.0, "entropy": 0.30428963899612427, "epoch": 0.09205928425111459, "frac_reward_zero_std": 0.0, "grad_norm": 3.2664194107055664, "learning_rate": 3.057575757575758e-06, "loss": -0.0461, "num_tokens": 5186427.0, "reward": 0.793583869934082, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8250000476837158, "reward_count_adherence_std": 0.0707106739282608, "reward_meter_mean": 0.9959465861320496, "reward_meter_std": 0.003204776206985116, "reward_repeat_penalty_mean": 0.9682539701461792, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.05434795841574669, "reward_total_composite_mean": 0.793583869934082, "reward_total_composite_std": 0.05434796214103699, "reward_total_mean": 0.793583869934082, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8250000476837158, "rewards/count_adherence/std": 0.0707106739282608, "rewards/meter/mean": 0.9959465861320496, "rewards/meter/std": 0.003204776206985116, "rewards/repeat_penalty/mean": 0.9682539701461792, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.793583869934082, "rewards/total_composite/std": 0.05434796214103699, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0047212839126587, "sampling/importance_sampling_ratio/min": 0.09027944505214691, "sampling/sampling_logp_difference/max": 2.4048454761505127, "sampling/sampling_logp_difference/mean": 0.03989114239811897, "step": 2292 }, { "clip_ratio/high_max": 0.017152261221781373, "clip_ratio/high_mean": 0.017152261221781373, "clip_ratio/low_mean": 0.010282430681400001, "clip_ratio/low_min": 0.010282430681400001, "clip_ratio/region_mean": 0.027434691903181374, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 363.875, "completions/mean_terminated_length": 363.875, "completions/min_length": 343.0, "completions/min_terminated_length": 343.0, "entropy": 0.26295676827430725, "epoch": 0.09209944973289955, "frac_reward_zero_std": 0.0, "grad_norm": 1.504547119140625, "learning_rate": 3.054545454545455e-06, "loss": 0.0327, "num_tokens": 5191210.0, "reward": 0.4955565929412842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6083333492279053, "reward_count_adherence_std": 0.0235702246427536, "reward_meter_mean": 0.9957020282745361, "reward_meter_std": 0.001336081069894135, "reward_repeat_penalty_mean": 0.8192294836044312, "reward_repeat_penalty_std": 0.07807844877243042, "reward_std": 0.04254494979977608, "reward_total_composite_mean": 0.4955565929412842, "reward_total_composite_std": 0.04254494607448578, "reward_total_mean": 0.4955565929412842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6083333492279053, "rewards/count_adherence/std": 0.0235702246427536, "rewards/meter/mean": 0.9957020282745361, "rewards/meter/std": 0.001336081069894135, "rewards/repeat_penalty/mean": 0.8192294836044312, "rewards/repeat_penalty/std": 0.07807844877243042, "rewards/total_composite/mean": 0.4955565929412842, "rewards/total_composite/std": 0.04254494607448578, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0060222148895264, "sampling/importance_sampling_ratio/min": 0.12191939353942871, "sampling/sampling_logp_difference/max": 2.1043951511383057, "sampling/sampling_logp_difference/mean": 0.032130178064107895, "step": 2293 }, { "clip_ratio/high_max": 0.014976657228544354, "clip_ratio/high_mean": 0.014976657228544354, "clip_ratio/low_mean": 0.00620853912550956, "clip_ratio/low_min": 0.00620853912550956, "clip_ratio/region_mean": 0.021185196354053915, "completions/clipped_ratio": 0.0, "completions/max_length": 326.0, "completions/max_terminated_length": 326.0, "completions/mean_length": 309.0, "completions/mean_terminated_length": 309.0, "completions/min_length": 287.0, "completions/min_terminated_length": 287.0, "entropy": 0.25325608998537064, "epoch": 0.0921396152146845, "frac_reward_zero_std": 0.0, "grad_norm": 1.8439388275146484, "learning_rate": 3.051515151515152e-06, "loss": -0.0027, "num_tokens": 5195522.0, "reward": 0.690649151802063, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7875000238418579, "reward_count_adherence_std": 0.0353553481400013, "reward_meter_mean": 0.99519944190979, "reward_meter_std": 0.003783464664593339, "reward_repeat_penalty_mean": 0.8807692527770996, "reward_repeat_penalty_std": 0.07008695602416992, "reward_std": 0.06761834770441055, "reward_total_composite_mean": 0.690649151802063, "reward_total_composite_std": 0.06761835515499115, "reward_total_mean": 0.690649151802063, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7875000238418579, "rewards/count_adherence/std": 0.0353553481400013, "rewards/meter/mean": 0.99519944190979, "rewards/meter/std": 0.003783464664593339, "rewards/repeat_penalty/mean": 0.8807692527770996, "rewards/repeat_penalty/std": 0.07008695602416992, "rewards/total_composite/mean": 0.690649151802063, "rewards/total_composite/std": 0.06761835515499115, "sampling/importance_sampling_ratio/max": 1.937671184539795, "sampling/importance_sampling_ratio/mean": 1.0094666481018066, "sampling/importance_sampling_ratio/min": 0.19834816455841064, "sampling/sampling_logp_difference/max": 1.6177313327789307, "sampling/sampling_logp_difference/mean": 0.03343411162495613, "step": 2294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.001179245300590992, "clip_ratio/low_min": 0.001179245300590992, "clip_ratio/region_mean": 0.001179245300590992, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 106.0, "completions/mean_terminated_length": 106.0, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.02729532145895064, "epoch": 0.09217978069646945, "frac_reward_zero_std": 0.0, "grad_norm": 3.1970479488372803, "learning_rate": 3.048484848484849e-06, "loss": 0.0008, "num_tokens": 5197722.0, "reward": 0.5567622184753418, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6495559215545654, "reward_meter_std": 0.1058802381157875, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09075447171926498, "reward_total_composite_mean": 0.5567622184753418, "reward_total_composite_std": 0.09075448662042618, "reward_total_mean": 0.5567622184753418, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6495559215545654, "rewards/meter/std": 0.1058802381157875, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5567622184753418, "rewards/total_composite/std": 0.09075448662042618, "sampling/importance_sampling_ratio/max": 1.2651506662368774, "sampling/importance_sampling_ratio/mean": 0.9996184706687927, "sampling/importance_sampling_ratio/min": 0.5777571797370911, "sampling/sampling_logp_difference/max": 0.5486016273498535, "sampling/sampling_logp_difference/mean": 0.003019496565684676, "step": 2295 }, { "clip_ratio/high_max": 0.0037878789007663727, "clip_ratio/high_mean": 0.0037878789007663727, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.00562611420173198, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.25, "completions/mean_terminated_length": 66.25, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.06576015800237656, "epoch": 0.09221994617825441, "frac_reward_zero_std": 0.0, "grad_norm": 2.241980791091919, "learning_rate": 3.045454545454546e-06, "loss": 0.005, "num_tokens": 5199420.0, "reward": 0.998041033744812, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998041033744812, "reward_meter_std": 8.582558075431734e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.582806185586378e-05, "reward_total_composite_mean": 0.998041033744812, "reward_total_composite_std": 8.582558075431734e-05, "reward_total_mean": 0.998041033744812, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998041033744812, "rewards/meter/std": 8.582558075431734e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998041033744812, "rewards/total_composite/std": 8.582558075431734e-05, "sampling/importance_sampling_ratio/max": 1.2699003219604492, "sampling/importance_sampling_ratio/mean": 0.99764084815979, "sampling/importance_sampling_ratio/min": 0.25096866488456726, "sampling/sampling_logp_difference/max": 1.3824272155761719, "sampling/sampling_logp_difference/mean": 0.014799090102314949, "step": 2296 }, { "clip_ratio/high_max": 0.0022727272007614374, "clip_ratio/high_mean": 0.0022727272007614374, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/region_mean": 0.004545454401522875, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 55.0, "completions/mean_terminated_length": 55.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.037125169299542904, "epoch": 0.09226011166003936, "frac_reward_zero_std": 0.0, "grad_norm": 0.5168666839599609, "learning_rate": 3.0424242424242427e-06, "loss": 0.0003, "num_tokens": 5201164.0, "reward": 0.9948955774307251, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948955774307251, "reward_meter_std": 1.7123466022894718e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.7113467038143426e-05, "reward_total_composite_mean": 0.9948955774307251, "reward_total_composite_std": 1.7123466022894718e-05, "reward_total_mean": 0.9948955774307251, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948955774307251, "rewards/meter/std": 1.7123466022894718e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948955774307251, "rewards/total_composite/std": 1.7123466022894718e-05, "sampling/importance_sampling_ratio/max": 1.3264281749725342, "sampling/importance_sampling_ratio/mean": 0.9979526400566101, "sampling/importance_sampling_ratio/min": 0.5306968092918396, "sampling/sampling_logp_difference/max": 0.6335644721984863, "sampling/sampling_logp_difference/mean": 0.0070167421363294125, "step": 2297 }, { "clip_ratio/high_max": 0.014611484948545694, "clip_ratio/high_mean": 0.014611484948545694, "clip_ratio/low_mean": 0.004934210330247879, "clip_ratio/low_min": 0.004934210330247879, "clip_ratio/region_mean": 0.019545695278793573, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 76.5, "completions/mean_terminated_length": 76.5, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.21845361031591892, "epoch": 0.09230027714182432, "frac_reward_zero_std": 0.0, "grad_norm": 2.7533559799194336, "learning_rate": 3.03939393939394e-06, "loss": -0.0044, "num_tokens": 5203008.0, "reward": 0.9956209659576416, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956209659576416, "reward_meter_std": 0.0032750838436186314, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00327507546171546, "reward_total_composite_mean": 0.9956209659576416, "reward_total_composite_std": 0.0032750838436186314, "reward_total_mean": 0.9956209659576416, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956209659576416, "rewards/meter/std": 0.0032750838436186314, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956209659576416, "rewards/total_composite/std": 0.0032750838436186314, "sampling/importance_sampling_ratio/max": 1.4816969633102417, "sampling/importance_sampling_ratio/mean": 1.0014814138412476, "sampling/importance_sampling_ratio/min": 0.3647305369377136, "sampling/sampling_logp_difference/max": 1.008596420288086, "sampling/sampling_logp_difference/mean": 0.0241053719073534, "step": 2298 }, { "clip_ratio/high_max": 0.02088787977118045, "clip_ratio/high_mean": 0.02088787977118045, "clip_ratio/low_mean": 0.01099004433490336, "clip_ratio/low_min": 0.01099004433490336, "clip_ratio/region_mean": 0.03187792410608381, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 136.375, "completions/mean_terminated_length": 136.375, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.24156404845416546, "epoch": 0.09234044262360927, "frac_reward_zero_std": 0.0, "grad_norm": 4.413851737976074, "learning_rate": 3.036363636363637e-06, "loss": 0.013, "num_tokens": 5205419.0, "reward": 0.754673421382904, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8257510662078857, "reward_meter_std": 0.3110801577568054, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.1079898476600647, "reward_std": 0.284267783164978, "reward_total_composite_mean": 0.754673421382904, "reward_total_composite_std": 0.2842678129673004, "reward_total_mean": 0.754673421382904, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8257510662078857, "rewards/meter/std": 0.3110801577568054, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.1079898476600647, "rewards/total_composite/mean": 0.754673421382904, "rewards/total_composite/std": 0.2842678129673004, "sampling/importance_sampling_ratio/max": 1.8313679695129395, "sampling/importance_sampling_ratio/mean": 1.0080208778381348, "sampling/importance_sampling_ratio/min": 0.1995229572057724, "sampling/sampling_logp_difference/max": 1.611825942993164, "sampling/sampling_logp_difference/mean": 0.02921801060438156, "step": 2299 }, { "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/region_mean": 0.0039100684225559235, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.625, "completions/mean_terminated_length": 65.625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.04499326320365071, "epoch": 0.09238060810539422, "frac_reward_zero_std": 0.0, "grad_norm": 2.978400230407715, "learning_rate": 3.0333333333333337e-06, "loss": -0.0163, "num_tokens": 5207160.0, "reward": 0.996369481086731, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.996369481086731, "reward_meter_std": 0.00487491674721241, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004874910227954388, "reward_total_composite_mean": 0.996369481086731, "reward_total_composite_std": 0.00487491674721241, "reward_total_mean": 0.996369481086731, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.996369481086731, "rewards/meter/std": 0.00487491674721241, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.996369481086731, "rewards/total_composite/std": 0.00487491674721241, "sampling/importance_sampling_ratio/max": 1.6583852767944336, "sampling/importance_sampling_ratio/mean": 1.004469871520996, "sampling/importance_sampling_ratio/min": 0.7272458076477051, "sampling/sampling_logp_difference/max": 0.5058443546295166, "sampling/sampling_logp_difference/mean": 0.007210928946733475, "step": 2300 }, { "epoch": 0.09238060810539422, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 324.3076923076923, "eval_completions/max_terminated_length": 324.3076923076923, "eval_completions/mean_length": 179.41346153846155, "eval_completions/mean_terminated_length": 179.41346153846155, "eval_completions/min_length": 60.46153846153846, "eval_completions/min_terminated_length": 60.46153846153846, "eval_entropy": 0.21972922522288102, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 5207160.0, "eval_reward": 0.5927730821646177, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.8744415503281814, "eval_reward_count_adherence_std": 0.1353673808849775, "eval_reward_meter_mean": 0.7395852987582867, "eval_reward_meter_std": 0.3918028657252972, "eval_reward_repeat_penalty_mean": 0.9028620398961581, "eval_reward_repeat_penalty_std": 0.10280066442031127, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.5927730821646177, "eval_reward_total_composite_std": 0.35657222683613116, "eval_reward_total_mean": 0.5927730821646177, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.8744415503281814, "eval_rewards/count_adherence/std": 0.1353673808849775, "eval_rewards/meter/mean": 0.7395852987582867, "eval_rewards/meter/std": 0.3918028657252972, "eval_rewards/repeat_penalty/mean": 0.9028620398961581, "eval_rewards/repeat_penalty/std": 0.10280066442031127, "eval_rewards/total_composite/mean": 0.5927730821646177, "eval_rewards/total_composite/std": 0.35657222683613116, "eval_runtime": 61.8872, "eval_samples_per_second": 1.68, "eval_sampling/importance_sampling_ratio/max": 1.405327586027292, "eval_sampling/importance_sampling_ratio/mean": 1.004779577255249, "eval_sampling/importance_sampling_ratio/min": 0.38055819043746364, "eval_sampling/sampling_logp_difference/max": 0.9916172761183518, "eval_sampling/sampling_logp_difference/mean": 0.02075868207388199, "eval_steps_per_second": 0.21, "step": 2300 }, { "clip_ratio/high_max": 0.01874305820092559, "clip_ratio/high_mean": 0.01874305820092559, "clip_ratio/low_mean": 0.008561643771827221, "clip_ratio/low_min": 0.008561643771827221, "clip_ratio/region_mean": 0.02730470197275281, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.21760859712958336, "epoch": 0.09242077358717918, "frac_reward_zero_std": 0.0, "grad_norm": 1.7726956605911255, "learning_rate": 3.0303030303030305e-06, "loss": -0.0022, "num_tokens": 5209084.0, "reward": 0.9988920092582703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988920092582703, "reward_meter_std": 0.0004005288355983794, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004005288355983794, "reward_total_composite_mean": 0.9988920092582703, "reward_total_composite_std": 0.0004005288355983794, "reward_total_mean": 0.9988920092582703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988920092582703, "rewards/meter/std": 0.0004005288355983794, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988920092582703, "rewards/total_composite/std": 0.0004005288355983794, "sampling/importance_sampling_ratio/max": 1.46310555934906, "sampling/importance_sampling_ratio/mean": 1.0048654079437256, "sampling/importance_sampling_ratio/min": 0.5579484701156616, "sampling/sampling_logp_difference/max": 0.5834887027740479, "sampling/sampling_logp_difference/mean": 0.023180123418569565, "step": 2301 }, { "clip_ratio/high_max": 0.025409106630831957, "clip_ratio/high_mean": 0.025409106630831957, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/region_mean": 0.02722070086747408, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 68.625, "completions/mean_terminated_length": 68.625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.26816077157855034, "epoch": 0.09246093906896413, "frac_reward_zero_std": 0.0, "grad_norm": 4.9054951667785645, "learning_rate": 3.0272727272727277e-06, "loss": 0.0094, "num_tokens": 5210937.0, "reward": 0.8321632742881775, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8321632742881775, "reward_meter_std": 0.27767401933670044, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.27767398953437805, "reward_total_composite_mean": 0.8321632742881775, "reward_total_composite_std": 0.27767401933670044, "reward_total_mean": 0.8321632742881775, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8321632742881775, "rewards/meter/std": 0.27767401933670044, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8321632742881775, "rewards/total_composite/std": 0.27767401933670044, "sampling/importance_sampling_ratio/max": 1.5462456941604614, "sampling/importance_sampling_ratio/mean": 1.0079941749572754, "sampling/importance_sampling_ratio/min": 0.22595034539699554, "sampling/sampling_logp_difference/max": 1.4874401092529297, "sampling/sampling_logp_difference/mean": 0.03458629176020622, "step": 2302 }, { "clip_ratio/high_max": 0.011514384881593287, "clip_ratio/high_mean": 0.011514384881593287, "clip_ratio/low_mean": 0.010549622820690274, "clip_ratio/low_min": 0.010549622820690274, "clip_ratio/region_mean": 0.02206400770228356, "completions/clipped_ratio": 0.0, "completions/max_length": 422.0, "completions/max_terminated_length": 422.0, "completions/mean_length": 387.875, "completions/mean_terminated_length": 387.875, "completions/min_length": 366.0, "completions/min_terminated_length": 366.0, "entropy": 0.2818215675652027, "epoch": 0.09250110455074909, "frac_reward_zero_std": 0.0, "grad_norm": 1.882811188697815, "learning_rate": 3.0242424242424246e-06, "loss": -0.0043, "num_tokens": 5215896.0, "reward": 0.5379288196563721, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.5882353186607361, "reward_count_adherence_std": 0.03144249692559242, "reward_meter_mean": 0.9971157908439636, "reward_meter_std": 0.001221620594151318, "reward_repeat_penalty_mean": 0.9170373678207397, "reward_repeat_penalty_std": 0.06482076644897461, "reward_std": 0.04898424819111824, "reward_total_composite_mean": 0.5379288196563721, "reward_total_composite_std": 0.04898425191640854, "reward_total_mean": 0.5379288196563721, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.5882353186607361, "rewards/count_adherence/std": 0.03144249692559242, "rewards/meter/mean": 0.9971157908439636, "rewards/meter/std": 0.001221620594151318, "rewards/repeat_penalty/mean": 0.9170373678207397, "rewards/repeat_penalty/std": 0.06482076644897461, "rewards/total_composite/mean": 0.5379288196563721, "rewards/total_composite/std": 0.04898425191640854, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0075607299804688, "sampling/importance_sampling_ratio/min": 0.25277209281921387, "sampling/sampling_logp_difference/max": 2.7829220294952393, "sampling/sampling_logp_difference/mean": 0.035993542522192, "step": 2303 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.004575719474814832, "epoch": 0.09254127003253404, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.0212121212121214e-06, "loss": 0.0, "num_tokens": 5217688.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0186870098114014, "sampling/importance_sampling_ratio/mean": 1.000432014465332, "sampling/importance_sampling_ratio/min": 0.9936742186546326, "sampling/sampling_logp_difference/max": 0.01851455122232437, "sampling/sampling_logp_difference/mean": 0.00046610343270003796, "step": 2304 }, { "clip_ratio/high_max": 0.014565107179805636, "clip_ratio/high_mean": 0.014565107179805636, "clip_ratio/low_mean": 0.014036171836778522, "clip_ratio/low_min": 0.014036171836778522, "clip_ratio/region_mean": 0.028601279016584158, "completions/clipped_ratio": 0.0, "completions/max_length": 192.0, "completions/max_terminated_length": 192.0, "completions/mean_length": 188.125, "completions/mean_terminated_length": 188.125, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.2865954786539078, "epoch": 0.092581435514319, "frac_reward_zero_std": 0.0, "grad_norm": 2.1037492752075195, "learning_rate": 3.0181818181818182e-06, "loss": -0.0071, "num_tokens": 5220865.0, "reward": 0.7725695371627808, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963282942771912, "reward_meter_std": 0.0020138672553002834, "reward_repeat_penalty_mean": 0.9305555820465088, "reward_repeat_penalty_std": 0.08266931027173996, "reward_std": 0.06809281557798386, "reward_total_composite_mean": 0.7725695371627808, "reward_total_composite_std": 0.06809280812740326, "reward_total_mean": 0.7725695371627808, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963282942771912, "rewards/meter/std": 0.0020138672553002834, "rewards/repeat_penalty/mean": 0.9305555820465088, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.7725695371627808, "rewards/total_composite/std": 0.06809280812740326, "sampling/importance_sampling_ratio/max": 1.9226046800613403, "sampling/importance_sampling_ratio/mean": 1.0070801973342896, "sampling/importance_sampling_ratio/min": 0.278171569108963, "sampling/sampling_logp_difference/max": 1.2795171737670898, "sampling/sampling_logp_difference/mean": 0.033207789063453674, "step": 2305 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.009295969794038683, "epoch": 0.09262160099610395, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.0151515151515155e-06, "loss": 0.0, "num_tokens": 5222265.0, "reward": 0.997984766960144, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997984766960144, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.997984766960144, "reward_total_composite_std": 0.0, "reward_total_mean": 0.997984766960144, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997984766960144, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997984766960144, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.025589942932129, "sampling/importance_sampling_ratio/mean": 1.0008283853530884, "sampling/importance_sampling_ratio/min": 0.9653224349021912, "sampling/sampling_logp_difference/max": 0.035293105989694595, "sampling/sampling_logp_difference/mean": 0.00116938806604594, "step": 2306 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.00621713837608695, "epoch": 0.0926617664778889, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.0121212121212123e-06, "loss": 0.0, "num_tokens": 5223977.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0361512899398804, "sampling/importance_sampling_ratio/mean": 1.0003576278686523, "sampling/importance_sampling_ratio/min": 0.9723794460296631, "sampling/sampling_logp_difference/max": 0.03551316633820534, "sampling/sampling_logp_difference/mean": 0.0005960848648101091, "step": 2307 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.007502294494770467, "epoch": 0.09270193195967386, "frac_reward_zero_std": 0.0, "grad_norm": 1.3243682384490967, "learning_rate": 3.009090909090909e-06, "loss": -0.0001, "num_tokens": 5225489.0, "reward": 0.9992917776107788, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992917776107788, "reward_meter_std": 1.824959326768294e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.82435705937678e-05, "reward_total_composite_mean": 0.9992917776107788, "reward_total_composite_std": 1.824959326768294e-05, "reward_total_mean": 0.9992917776107788, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992917776107788, "rewards/meter/std": 1.824959326768294e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992917776107788, "rewards/total_composite/std": 1.824959326768294e-05, "sampling/importance_sampling_ratio/max": 1.0328819751739502, "sampling/importance_sampling_ratio/mean": 0.9991614818572998, "sampling/importance_sampling_ratio/min": 0.7382054924964905, "sampling/sampling_logp_difference/max": 0.30353307723999023, "sampling/sampling_logp_difference/mean": 0.0022324356250464916, "step": 2308 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.125, "completions/mean_terminated_length": 64.125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.013516030623577535, "epoch": 0.09274209744145881, "frac_reward_zero_std": 0.0, "grad_norm": 1.3253436088562012, "learning_rate": 3.0060606060606064e-06, "loss": 0.0018, "num_tokens": 5227402.0, "reward": 0.9992837905883789, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992837905883789, "reward_meter_std": 7.721914880676195e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.722593727521598e-05, "reward_total_composite_mean": 0.9992837905883789, "reward_total_composite_std": 7.721914880676195e-05, "reward_total_mean": 0.9992837905883789, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992837905883789, "rewards/meter/std": 7.721914880676195e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992837905883789, "rewards/total_composite/std": 7.721914880676195e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002130150794983, "sampling/importance_sampling_ratio/min": 0.7730492949485779, "sampling/sampling_logp_difference/max": 1.3222827911376953, "sampling/sampling_logp_difference/mean": 0.004359652288258076, "step": 2309 }, { "clip_ratio/high_max": 0.006818181602284312, "clip_ratio/high_mean": 0.006818181602284312, "clip_ratio/low_mean": 0.004545454401522875, "clip_ratio/low_min": 0.004545454401522875, "clip_ratio/region_mean": 0.011363636003807187, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 55.0, "completions/mean_terminated_length": 55.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.03565134946256876, "epoch": 0.09278226292324376, "frac_reward_zero_std": 0.0, "grad_norm": 0.4977571368217468, "learning_rate": 3.0030303030303032e-06, "loss": -0.001, "num_tokens": 5229178.0, "reward": 0.9948618412017822, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948618412017822, "reward_meter_std": 9.404453885508701e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.405182208865881e-05, "reward_total_composite_mean": 0.9948618412017822, "reward_total_composite_std": 9.404453885508701e-05, "reward_total_mean": 0.9948618412017822, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948618412017822, "rewards/meter/std": 9.404453885508701e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948618412017822, "rewards/total_composite/std": 9.404453885508701e-05, "sampling/importance_sampling_ratio/max": 1.473840355873108, "sampling/importance_sampling_ratio/mean": 1.000832200050354, "sampling/importance_sampling_ratio/min": 0.5851462483406067, "sampling/sampling_logp_difference/max": 0.535893440246582, "sampling/sampling_logp_difference/mean": 0.008051915094256401, "step": 2310 }, { "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/region_mean": 0.006818181602284312, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 55.0, "completions/mean_terminated_length": 55.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.032598731108009815, "epoch": 0.09282242840502872, "frac_reward_zero_std": 0.0, "grad_norm": 0.3652920722961426, "learning_rate": 3e-06, "loss": -0.0002, "num_tokens": 5230866.0, "reward": 0.9948963522911072, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948963522911072, "reward_meter_std": 1.816852636693511e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.817071097320877e-05, "reward_total_composite_mean": 0.9948963522911072, "reward_total_composite_std": 1.816852636693511e-05, "reward_total_mean": 0.9948963522911072, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948963522911072, "rewards/meter/std": 1.816852636693511e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948963522911072, "rewards/total_composite/std": 1.816852636693511e-05, "sampling/importance_sampling_ratio/max": 1.3571921586990356, "sampling/importance_sampling_ratio/mean": 0.999363124370575, "sampling/importance_sampling_ratio/min": 0.7490553259849548, "sampling/sampling_logp_difference/max": 0.3054179549217224, "sampling/sampling_logp_difference/mean": 0.00472200708463788, "step": 2311 }, { "clip_ratio/high_max": 0.006349399336613715, "clip_ratio/high_mean": 0.006349399336613715, "clip_ratio/low_mean": 0.011666666949167848, "clip_ratio/low_min": 0.011666666949167848, "clip_ratio/region_mean": 0.018016066285781562, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 76.5, "completions/mean_terminated_length": 76.5, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.18640629388391972, "epoch": 0.09286259388681367, "frac_reward_zero_std": 0.0, "grad_norm": 2.4858875274658203, "learning_rate": 2.996969696969697e-06, "loss": -0.0133, "num_tokens": 5232710.0, "reward": 0.9963726997375488, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963726997375488, "reward_meter_std": 0.000908972229808569, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009089733357541263, "reward_total_composite_mean": 0.9963726997375488, "reward_total_composite_std": 0.000908972229808569, "reward_total_mean": 0.9963726997375488, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963726997375488, "rewards/meter/std": 0.000908972229808569, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963726997375488, "rewards/total_composite/std": 0.000908972229808569, "sampling/importance_sampling_ratio/max": 1.9459387063980103, "sampling/importance_sampling_ratio/mean": 1.0097090005874634, "sampling/importance_sampling_ratio/min": 0.32876816391944885, "sampling/sampling_logp_difference/max": 1.1124024391174316, "sampling/sampling_logp_difference/mean": 0.024638459086418152, "step": 2312 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 54.875, "completions/mean_terminated_length": 54.875, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.02833111328072846, "epoch": 0.09290275936859863, "frac_reward_zero_std": 0.0, "grad_norm": 1.909967303276062, "learning_rate": 2.993939393939394e-06, "loss": 0.001, "num_tokens": 5234365.0, "reward": 0.9950602054595947, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9950602054595947, "reward_meter_std": 0.0005074667278677225, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005074667278677225, "reward_total_composite_mean": 0.9950602054595947, "reward_total_composite_std": 0.0005074667278677225, "reward_total_mean": 0.9950602054595947, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9950602054595947, "rewards/meter/std": 0.0005074667278677225, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950602054595947, "rewards/total_composite/std": 0.0005074667278677225, "sampling/importance_sampling_ratio/max": 1.1668245792388916, "sampling/importance_sampling_ratio/mean": 0.9969737529754639, "sampling/importance_sampling_ratio/min": 0.47267839312553406, "sampling/sampling_logp_difference/max": 0.7493400573730469, "sampling/sampling_logp_difference/mean": 0.007333268411457539, "step": 2313 }, { "clip_ratio/high_max": 0.0221836578566581, "clip_ratio/high_mean": 0.0221836578566581, "clip_ratio/low_mean": 0.0008389261784031987, "clip_ratio/low_min": 0.0008389261784031987, "clip_ratio/region_mean": 0.0230225840350613, "completions/clipped_ratio": 0.0, "completions/max_length": 154.0, "completions/max_terminated_length": 154.0, "completions/mean_length": 151.75, "completions/mean_terminated_length": 151.75, "completions/min_length": 149.0, "completions/min_terminated_length": 149.0, "entropy": 0.21811402402818203, "epoch": 0.09294292485038358, "frac_reward_zero_std": 0.0, "grad_norm": 1.915865182876587, "learning_rate": 2.990909090909091e-06, "loss": -0.005, "num_tokens": 5237243.0, "reward": 0.9612998962402344, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9968797564506531, "reward_meter_std": 0.0010569265577942133, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.10101525485515594, "reward_std": 0.10089994221925735, "reward_total_composite_mean": 0.9612998962402344, "reward_total_composite_std": 0.10089994966983795, "reward_total_mean": 0.9612998962402344, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9968797564506531, "rewards/meter/std": 0.0010569265577942133, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.10101525485515594, "rewards/total_composite/mean": 0.9612998962402344, "rewards/total_composite/std": 0.10089994966983795, "sampling/importance_sampling_ratio/max": 1.6189171075820923, "sampling/importance_sampling_ratio/mean": 1.0044018030166626, "sampling/importance_sampling_ratio/min": 0.29246416687965393, "sampling/sampling_logp_difference/max": 1.2294130325317383, "sampling/sampling_logp_difference/mean": 0.027893707156181335, "step": 2314 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.005022522644139826, "clip_ratio/low_min": 0.005022522644139826, "clip_ratio/region_mean": 0.005022522644139826, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 74.875, "completions/mean_terminated_length": 74.875, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.21749908849596977, "epoch": 0.09298309033216853, "frac_reward_zero_std": 0.0, "grad_norm": 2.175835132598877, "learning_rate": 2.987878787878788e-06, "loss": -0.001, "num_tokens": 5239130.0, "reward": 0.9990985989570618, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990985989570618, "reward_meter_std": 0.00011346396786393598, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011346933752065524, "reward_total_composite_mean": 0.9990985989570618, "reward_total_composite_std": 0.00011346396786393598, "reward_total_mean": 0.9990985989570618, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990985989570618, "rewards/meter/std": 0.00011346396786393598, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990985989570618, "rewards/total_composite/std": 0.00011346396786393598, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0061297416687012, "sampling/importance_sampling_ratio/min": 0.32103458046913147, "sampling/sampling_logp_difference/max": 1.1362063884735107, "sampling/sampling_logp_difference/mean": 0.02791663445532322, "step": 2315 }, { "clip_ratio/high_max": 0.0029296875, "clip_ratio/high_mean": 0.0029296875, "clip_ratio/low_mean": 0.0049212598241865635, "clip_ratio/low_min": 0.0049212598241865635, "clip_ratio/region_mean": 0.007850947324186563, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 128.0, "completions/mean_terminated_length": 128.0, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.0738038350827992, "epoch": 0.09302325581395349, "frac_reward_zero_std": 0.0, "grad_norm": 4.3104939460754395, "learning_rate": 2.984848484848485e-06, "loss": 0.0011, "num_tokens": 5241562.0, "reward": 0.7751978635787964, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9043974876403809, "reward_meter_std": 0.2645852863788605, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.22678740322589874, "reward_total_composite_mean": 0.7751978635787964, "reward_total_composite_std": 0.22678740322589874, "reward_total_mean": 0.7751978635787964, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9043974876403809, "rewards/meter/std": 0.2645852863788605, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7751978635787964, "rewards/total_composite/std": 0.22678740322589874, "sampling/importance_sampling_ratio/max": 1.5058414936065674, "sampling/importance_sampling_ratio/mean": 1.003368854522705, "sampling/importance_sampling_ratio/min": 0.3948022723197937, "sampling/sampling_logp_difference/max": 0.9293702840805054, "sampling/sampling_logp_difference/mean": 0.01084654126316309, "step": 2316 }, { "clip_ratio/high_max": 0.009331883396953344, "clip_ratio/high_mean": 0.009331883396953344, "clip_ratio/low_mean": 0.012263738899491727, "clip_ratio/low_min": 0.012263738899491727, "clip_ratio/region_mean": 0.02159562229644507, "completions/clipped_ratio": 0.0, "completions/max_length": 450.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 424.125, "completions/mean_terminated_length": 424.125, "completions/min_length": 401.0, "completions/min_terminated_length": 401.0, "entropy": 0.2362643126398325, "epoch": 0.09306342129573844, "frac_reward_zero_std": 0.0, "grad_norm": 1.892166256904602, "learning_rate": 2.981818181818182e-06, "loss": -0.0076, "num_tokens": 5246963.0, "reward": 0.5997676849365234, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6447368264198303, "reward_count_adherence_std": 0.024363704025745392, "reward_meter_mean": 0.992563009262085, "reward_meter_std": 0.0028387887869030237, "reward_repeat_penalty_mean": 0.9372192025184631, "reward_repeat_penalty_std": 0.03141416236758232, "reward_std": 0.030326958745718002, "reward_total_composite_mean": 0.5997676849365234, "reward_total_composite_std": 0.030326971784234047, "reward_total_mean": 0.5997676849365234, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6447368264198303, "rewards/count_adherence/std": 0.024363704025745392, "rewards/meter/mean": 0.992563009262085, "rewards/meter/std": 0.0028387887869030237, "rewards/repeat_penalty/mean": 0.9372192025184631, "rewards/repeat_penalty/std": 0.03141416236758232, "rewards/total_composite/mean": 0.5997676849365234, "rewards/total_composite/std": 0.030326971784234047, "sampling/importance_sampling_ratio/max": 1.956535816192627, "sampling/importance_sampling_ratio/mean": 1.0031834840774536, "sampling/importance_sampling_ratio/min": 0.1540173441171646, "sampling/sampling_logp_difference/max": 1.870690107345581, "sampling/sampling_logp_difference/mean": 0.0303457360714674, "step": 2317 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/region_mean": 0.007699597394093871, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.037344205658882856, "epoch": 0.0931035867775234, "frac_reward_zero_std": 0.0, "grad_norm": 1.0140128135681152, "learning_rate": 2.9787878787878787e-06, "loss": -0.0195, "num_tokens": 5248971.0, "reward": 0.9957113265991211, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957113265991211, "reward_meter_std": 0.006800828501582146, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0068008191883563995, "reward_total_composite_mean": 0.9957113265991211, "reward_total_composite_std": 0.006800828501582146, "reward_total_mean": 0.9957113265991211, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957113265991211, "rewards/meter/std": 0.006800828501582146, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957113265991211, "rewards/total_composite/std": 0.006800828501582146, "sampling/importance_sampling_ratio/max": 1.1265363693237305, "sampling/importance_sampling_ratio/mean": 0.9992537498474121, "sampling/importance_sampling_ratio/min": 0.3114334046840668, "sampling/sampling_logp_difference/max": 1.166569709777832, "sampling/sampling_logp_difference/mean": 0.006797808688133955, "step": 2318 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0041495013574603945, "epoch": 0.09314375225930835, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.9757575757575756e-06, "loss": 0.0, "num_tokens": 5250619.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0146803855895996, "sampling/importance_sampling_ratio/mean": 1.0004198551177979, "sampling/importance_sampling_ratio/min": 0.9997897744178772, "sampling/sampling_logp_difference/max": 0.014573629945516586, "sampling/sampling_logp_difference/mean": 0.00042203269549645483, "step": 2319 }, { "clip_ratio/high_max": 0.0024999999441206455, "clip_ratio/high_mean": 0.0024999999441206455, "clip_ratio/low_mean": 0.01485577883431688, "clip_ratio/low_min": 0.01485577883431688, "clip_ratio/region_mean": 0.017355778778437525, "completions/clipped_ratio": 0.0, "completions/max_length": 204.0, "completions/max_terminated_length": 204.0, "completions/mean_length": 202.25, "completions/mean_terminated_length": 202.25, "completions/min_length": 200.0, "completions/min_terminated_length": 200.0, "entropy": 0.12941910233348608, "epoch": 0.0931839177410933, "frac_reward_zero_std": 0.0, "grad_norm": 1.6090915203094482, "learning_rate": 2.9727272727272733e-06, "loss": 0.0058, "num_tokens": 5253749.0, "reward": 0.7856208682060242, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957519769668579, "reward_meter_std": 0.004873792175203562, "reward_repeat_penalty_mean": 0.9204546213150024, "reward_repeat_penalty_std": 0.03214120864868164, "reward_std": 0.028088262304663658, "reward_total_composite_mean": 0.7856208682060242, "reward_total_composite_std": 0.02808825671672821, "reward_total_mean": 0.7856208682060242, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957519769668579, "rewards/meter/std": 0.004873792175203562, "rewards/repeat_penalty/mean": 0.9204546213150024, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.7856208682060242, "rewards/total_composite/std": 0.02808825671672821, "sampling/importance_sampling_ratio/max": 1.7268435955047607, "sampling/importance_sampling_ratio/mean": 0.9995094537734985, "sampling/importance_sampling_ratio/min": 0.04189267382025719, "sampling/sampling_logp_difference/max": 3.1726443767547607, "sampling/sampling_logp_difference/mean": 0.02219260297715664, "step": 2320 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.004189703788142651, "epoch": 0.09322408322287826, "frac_reward_zero_std": 0.0, "grad_norm": 0.006876571103930473, "learning_rate": 2.96969696969697e-06, "loss": -0.0002, "num_tokens": 5255469.0, "reward": 0.7876332998275757, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7876332998275757, "reward_meter_std": 1.5573279597447254e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.558229632792063e-05, "reward_total_composite_mean": 0.7876332998275757, "reward_total_composite_std": 1.5573279597447254e-05, "reward_total_mean": 0.7876332998275757, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7876332998275757, "rewards/meter/std": 1.5573279597447254e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876332998275757, "rewards/total_composite/std": 1.5573279597447254e-05, "sampling/importance_sampling_ratio/max": 1.0165002346038818, "sampling/importance_sampling_ratio/mean": 0.9997325539588928, "sampling/importance_sampling_ratio/min": 0.6743254065513611, "sampling/sampling_logp_difference/max": 0.3940424919128418, "sampling/sampling_logp_difference/mean": 0.0014006540877744555, "step": 2321 }, { "clip_ratio/high_max": 0.04014339251443744, "clip_ratio/high_mean": 0.04014339251443744, "clip_ratio/low_mean": 0.032118565402925014, "clip_ratio/low_min": 0.032118565402925014, "clip_ratio/region_mean": 0.07226195791736245, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 67.875, "completions/mean_terminated_length": 67.875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.3680881727486849, "epoch": 0.09326424870466321, "frac_reward_zero_std": 0.0, "grad_norm": 7.9188666343688965, "learning_rate": 2.9666666666666673e-06, "loss": 0.0375, "num_tokens": 5257324.0, "reward": 0.5824298858642578, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7128692269325256, "reward_meter_std": 0.2854415774345398, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.14880475401878357, "reward_std": 0.28861746191978455, "reward_total_composite_mean": 0.5824298858642578, "reward_total_composite_std": 0.28861746191978455, "reward_total_mean": 0.5824298858642578, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7128692269325256, "rewards/meter/std": 0.2854415774345398, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.14880475401878357, "rewards/total_composite/mean": 0.5824298858642578, "rewards/total_composite/std": 0.28861746191978455, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028222799301147, "sampling/importance_sampling_ratio/min": 0.3399814963340759, "sampling/sampling_logp_difference/max": 1.0788640975952148, "sampling/sampling_logp_difference/mean": 0.07370011508464813, "step": 2322 }, { "clip_ratio/high_max": 0.001923076924867928, "clip_ratio/high_mean": 0.001923076924867928, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.001923076924867928, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.125, "completions/mean_terminated_length": 64.125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.009595987561624497, "epoch": 0.09330441418644816, "frac_reward_zero_std": 0.0, "grad_norm": 0.16411831974983215, "learning_rate": 2.963636363636364e-06, "loss": 0.0003, "num_tokens": 5259053.0, "reward": 0.9993196725845337, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993196725845337, "reward_meter_std": 1.4266728612710722e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.4269724488258362e-05, "reward_total_composite_mean": 0.9993196725845337, "reward_total_composite_std": 1.4266728612710722e-05, "reward_total_mean": 0.9993196725845337, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993196725845337, "rewards/meter/std": 1.4266728612710722e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993196725845337, "rewards/total_composite/std": 1.4266728612710722e-05, "sampling/importance_sampling_ratio/max": 1.0394550561904907, "sampling/importance_sampling_ratio/mean": 0.9990805387496948, "sampling/importance_sampling_ratio/min": 0.4420400857925415, "sampling/sampling_logp_difference/max": 0.8163547515869141, "sampling/sampling_logp_difference/mean": 0.0027244146913290024, "step": 2323 }, { "clip_ratio/high_max": 0.021577789448201656, "clip_ratio/high_mean": 0.021577789448201656, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/region_mean": 0.025888134259730577, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 115.375, "completions/mean_terminated_length": 115.375, "completions/min_length": 112.0, "completions/min_terminated_length": 112.0, "entropy": 0.2400868460536003, "epoch": 0.09334457966823312, "frac_reward_zero_std": 0.0, "grad_norm": 3.402935028076172, "learning_rate": 2.960606060606061e-06, "loss": 0.0001, "num_tokens": 5261376.0, "reward": 0.9961607456207275, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961607456207275, "reward_meter_std": 0.003092888044193387, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0030928829219192266, "reward_total_composite_mean": 0.9961607456207275, "reward_total_composite_std": 0.003092888044193387, "reward_total_mean": 0.9961607456207275, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961607456207275, "rewards/meter/std": 0.003092888044193387, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961607456207275, "rewards/total_composite/std": 0.003092888044193387, "sampling/importance_sampling_ratio/max": 1.6382875442504883, "sampling/importance_sampling_ratio/mean": 1.0031054019927979, "sampling/importance_sampling_ratio/min": 0.2535265386104584, "sampling/sampling_logp_difference/max": 1.3722867965698242, "sampling/sampling_logp_difference/mean": 0.027808500453829765, "step": 2324 }, { "clip_ratio/high_max": 0.02963537711184472, "clip_ratio/high_mean": 0.02963537711184472, "clip_ratio/low_mean": 0.009615384973585606, "clip_ratio/low_min": 0.009615384973585606, "clip_ratio/region_mean": 0.039250762085430324, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.375, "completions/mean_terminated_length": 67.375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.3584112077951431, "epoch": 0.09338474515001807, "frac_reward_zero_std": 0.0, "grad_norm": 4.780300140380859, "learning_rate": 2.957575757575758e-06, "loss": 0.0074, "num_tokens": 5263331.0, "reward": 0.9977126121520996, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977126121520996, "reward_meter_std": 0.0028336539398878813, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0028336546383798122, "reward_total_composite_mean": 0.9977126121520996, "reward_total_composite_std": 0.0028336539398878813, "reward_total_mean": 0.9977126121520996, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977126121520996, "rewards/meter/std": 0.0028336539398878813, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977126121520996, "rewards/total_composite/std": 0.0028336539398878813, "sampling/importance_sampling_ratio/max": 1.6862565279006958, "sampling/importance_sampling_ratio/mean": 1.0002453327178955, "sampling/importance_sampling_ratio/min": 0.31873372197151184, "sampling/sampling_logp_difference/max": 1.1433992385864258, "sampling/sampling_logp_difference/mean": 0.049458228051662445, "step": 2325 }, { "clip_ratio/high_max": 0.011138874455355108, "clip_ratio/high_mean": 0.011138874455355108, "clip_ratio/low_mean": 0.0033333334140479565, "clip_ratio/low_min": 0.0033333334140479565, "clip_ratio/region_mean": 0.014472207869403064, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 78.0, "completions/mean_terminated_length": 78.0, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.16736672818660736, "epoch": 0.09342491063180303, "frac_reward_zero_std": 0.0, "grad_norm": 3.8735227584838867, "learning_rate": 2.954545454545455e-06, "loss": -0.0182, "num_tokens": 5265267.0, "reward": 0.9950919151306152, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9950919151306152, "reward_meter_std": 0.007245616987347603, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007245623506605625, "reward_total_composite_mean": 0.9950919151306152, "reward_total_composite_std": 0.007245616987347603, "reward_total_mean": 0.9950919151306152, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9950919151306152, "rewards/meter/std": 0.007245616987347603, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950919151306152, "rewards/total_composite/std": 0.007245616987347603, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0039349794387817, "sampling/importance_sampling_ratio/min": 0.4386661946773529, "sampling/sampling_logp_difference/max": 0.8240165710449219, "sampling/sampling_logp_difference/mean": 0.01848999224603176, "step": 2326 }, { "clip_ratio/high_max": 0.009627727791666985, "clip_ratio/high_mean": 0.009627727791666985, "clip_ratio/low_mean": 0.00937500037252903, "clip_ratio/low_min": 0.00937500037252903, "clip_ratio/region_mean": 0.019002728164196014, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 39.0, "completions/mean_terminated_length": 39.0, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.19926494918763638, "epoch": 0.09346507611358798, "frac_reward_zero_std": 0.0, "grad_norm": 7.240959167480469, "learning_rate": 2.951515151515152e-06, "loss": -0.0017, "num_tokens": 5266851.0, "reward": 0.9987367391586304, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987367391586304, "reward_meter_std": 0.0013684039004147053, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0013684096047654748, "reward_total_composite_mean": 0.9987367391586304, "reward_total_composite_std": 0.0013684039004147053, "reward_total_mean": 0.9987367391586304, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987367391586304, "rewards/meter/std": 0.0013684039004147053, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987367391586304, "rewards/total_composite/std": 0.0013684039004147053, "sampling/importance_sampling_ratio/max": 1.3673170804977417, "sampling/importance_sampling_ratio/mean": 1.003873348236084, "sampling/importance_sampling_ratio/min": 0.42664220929145813, "sampling/sampling_logp_difference/max": 0.8518095016479492, "sampling/sampling_logp_difference/mean": 0.032071422785520554, "step": 2327 }, { "clip_ratio/high_max": 0.0031250000465661287, "clip_ratio/high_mean": 0.0031250000465661287, "clip_ratio/low_mean": 0.0007812500116415322, "clip_ratio/low_min": 0.0007812500116415322, "clip_ratio/region_mean": 0.003906250058207661, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 159.625, "completions/mean_terminated_length": 159.625, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.060957128182053566, "epoch": 0.09350524159537293, "frac_reward_zero_std": 0.0, "grad_norm": 0.9024956822395325, "learning_rate": 2.9484848484848488e-06, "loss": 0.0007, "num_tokens": 5269464.0, "reward": 0.8176930546760559, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978625774383545, "reward_meter_std": 3.399623528821394e-05, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.05738433822989464, "reward_total_composite_mean": 0.8176930546760559, "reward_total_composite_std": 0.057384345680475235, "reward_total_mean": 0.8176930546760559, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978625774383545, "rewards/meter/std": 3.399623528821394e-05, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.8176930546760559, "rewards/total_composite/std": 0.057384345680475235, "sampling/importance_sampling_ratio/max": 1.274688720703125, "sampling/importance_sampling_ratio/mean": 0.9999021291732788, "sampling/importance_sampling_ratio/min": 0.3360353410243988, "sampling/sampling_logp_difference/max": 1.0905389785766602, "sampling/sampling_logp_difference/mean": 0.010679478757083416, "step": 2328 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.013287583249621093, "clip_ratio/low_min": 0.013287583249621093, "clip_ratio/region_mean": 0.017018926446326077, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.1675149966031313, "epoch": 0.09354540707715789, "frac_reward_zero_std": 0.0, "grad_norm": 3.704277992248535, "learning_rate": 2.9454545454545456e-06, "loss": 0.0043, "num_tokens": 5271316.0, "reward": 0.9816258549690247, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9816258549690247, "reward_meter_std": 0.004315865226089954, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004315854981541634, "reward_total_composite_mean": 0.9816258549690247, "reward_total_composite_std": 0.004315865226089954, "reward_total_mean": 0.9816258549690247, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9816258549690247, "rewards/meter/std": 0.004315865226089954, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9816258549690247, "rewards/total_composite/std": 0.004315865226089954, "sampling/importance_sampling_ratio/max": 1.3998769521713257, "sampling/importance_sampling_ratio/mean": 1.0040234327316284, "sampling/importance_sampling_ratio/min": 0.3291874825954437, "sampling/sampling_logp_difference/max": 1.1111278533935547, "sampling/sampling_logp_difference/mean": 0.022112296894192696, "step": 2329 }, { "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.005542142200283706, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.875, "completions/mean_terminated_length": 67.875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.0950354440137744, "epoch": 0.09358557255894284, "frac_reward_zero_std": 0.0, "grad_norm": 4.507577419281006, "learning_rate": 2.942424242424243e-06, "loss": 0.001, "num_tokens": 5272995.0, "reward": 0.9815051555633545, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9815051555633545, "reward_meter_std": 0.004705195315182209, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004705206956714392, "reward_total_composite_mean": 0.9815051555633545, "reward_total_composite_std": 0.004705195315182209, "reward_total_mean": 0.9815051555633545, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9815051555633545, "rewards/meter/std": 0.004705195315182209, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9815051555633545, "rewards/total_composite/std": 0.004705195315182209, "sampling/importance_sampling_ratio/max": 1.3674873113632202, "sampling/importance_sampling_ratio/mean": 1.0061925649642944, "sampling/importance_sampling_ratio/min": 0.5306636691093445, "sampling/sampling_logp_difference/max": 0.6336269378662109, "sampling/sampling_logp_difference/mean": 0.015049697831273079, "step": 2330 }, { "clip_ratio/high_max": 0.018828062573447824, "clip_ratio/high_mean": 0.018828062573447824, "clip_ratio/low_mean": 0.020951705053448677, "clip_ratio/low_min": 0.020951705053448677, "clip_ratio/region_mean": 0.0397797676268965, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.75, "completions/mean_terminated_length": 65.75, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.3256809823215008, "epoch": 0.0936257380407278, "frac_reward_zero_std": 0.0, "grad_norm": 3.0374279022216797, "learning_rate": 2.9393939393939397e-06, "loss": 0.0002, "num_tokens": 5274705.0, "reward": 0.9988436698913574, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988436698913574, "reward_meter_std": 0.00024908926570788026, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000249094155151397, "reward_total_composite_mean": 0.9988436698913574, "reward_total_composite_std": 0.00024908926570788026, "reward_total_mean": 0.9988436698913574, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988436698913574, "rewards/meter/std": 0.00024908926570788026, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988436698913574, "rewards/total_composite/std": 0.00024908926570788026, "sampling/importance_sampling_ratio/max": 1.7954013347625732, "sampling/importance_sampling_ratio/mean": 1.0106713771820068, "sampling/importance_sampling_ratio/min": 0.3650510609149933, "sampling/sampling_logp_difference/max": 1.0077180862426758, "sampling/sampling_logp_difference/mean": 0.04332128167152405, "step": 2331 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.005597014795057476, "clip_ratio/low_min": 0.005597014795057476, "clip_ratio/region_mean": 0.005597014795057476, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.05256783217191696, "epoch": 0.09366590352251275, "frac_reward_zero_std": 0.0, "grad_norm": 0.5694947242736816, "learning_rate": 2.9363636363636365e-06, "loss": -0.0003, "num_tokens": 5276497.0, "reward": 0.998113214969635, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998113214969635, "reward_meter_std": 3.099795139860362e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.100236426689662e-05, "reward_total_composite_mean": 0.998113214969635, "reward_total_composite_std": 3.099795139860362e-05, "reward_total_mean": 0.998113214969635, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998113214969635, "rewards/meter/std": 3.099795139860362e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998113214969635, "rewards/total_composite/std": 3.099795139860362e-05, "sampling/importance_sampling_ratio/max": 1.6667267084121704, "sampling/importance_sampling_ratio/mean": 1.002618670463562, "sampling/importance_sampling_ratio/min": 0.4737045168876648, "sampling/sampling_logp_difference/max": 0.7471715211868286, "sampling/sampling_logp_difference/mean": 0.007214725017547607, "step": 2332 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0035651897196657956, "epoch": 0.0937060690042977, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.9333333333333338e-06, "loss": 0.0, "num_tokens": 5278425.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0177748203277588, "sampling/importance_sampling_ratio/mean": 1.0003622770309448, "sampling/importance_sampling_ratio/min": 0.9928526878356934, "sampling/sampling_logp_difference/max": 0.01761876977980137, "sampling/sampling_logp_difference/mean": 0.0004051509313285351, "step": 2333 }, { "clip_ratio/high_max": 0.011166593292728066, "clip_ratio/high_mean": 0.011166593292728066, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.011166593292728066, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0710354926995933, "epoch": 0.09374623448608266, "frac_reward_zero_std": 0.0, "grad_norm": 8.575316429138184, "learning_rate": 2.9303030303030306e-06, "loss": -0.0067, "num_tokens": 5280271.0, "reward": 0.9969208836555481, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969208836555481, "reward_meter_std": 0.0032654430251568556, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003265430685132742, "reward_total_composite_mean": 0.9969208836555481, "reward_total_composite_std": 0.0032654430251568556, "reward_total_mean": 0.9969208836555481, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969208836555481, "rewards/meter/std": 0.0032654430251568556, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969208836555481, "rewards/total_composite/std": 0.0032654430251568556, "sampling/importance_sampling_ratio/max": 1.4511247873306274, "sampling/importance_sampling_ratio/mean": 0.9968401193618774, "sampling/importance_sampling_ratio/min": 0.29098066687583923, "sampling/sampling_logp_difference/max": 1.2344985008239746, "sampling/sampling_logp_difference/mean": 0.012270020321011543, "step": 2334 }, { "clip_ratio/high_max": 0.0022727272007614374, "clip_ratio/high_mean": 0.0022727272007614374, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/region_mean": 0.004545454401522875, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 55.0, "completions/mean_terminated_length": 55.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.022867268649861217, "epoch": 0.09378639996786761, "frac_reward_zero_std": 0.0, "grad_norm": 2.4751315116882324, "learning_rate": 2.9272727272727274e-06, "loss": 0.0011, "num_tokens": 5281911.0, "reward": 0.9950971603393555, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9950971603393555, "reward_meter_std": 0.0004900125786662102, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004900065250694752, "reward_total_composite_mean": 0.9950971603393555, "reward_total_composite_std": 0.0004900125786662102, "reward_total_mean": 0.9950971603393555, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9950971603393555, "rewards/meter/std": 0.0004900125786662102, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950971603393555, "rewards/total_composite/std": 0.0004900125786662102, "sampling/importance_sampling_ratio/max": 1.2244776487350464, "sampling/importance_sampling_ratio/mean": 0.9981791973114014, "sampling/importance_sampling_ratio/min": 0.6740844249725342, "sampling/sampling_logp_difference/max": 0.3944000005722046, "sampling/sampling_logp_difference/mean": 0.005754906218498945, "step": 2335 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.04221090069040656, "epoch": 0.09382656544965257, "frac_reward_zero_std": 0.0, "grad_norm": 2.559536933898926, "learning_rate": 2.9242424242424243e-06, "loss": 0.0022, "num_tokens": 5283616.0, "reward": 0.9981128573417664, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981128573417664, "reward_meter_std": 5.023560152039863e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.024260462960228e-05, "reward_total_composite_mean": 0.9981128573417664, "reward_total_composite_std": 5.023560152039863e-05, "reward_total_mean": 0.9981128573417664, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981128573417664, "rewards/meter/std": 5.023560152039863e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981128573417664, "rewards/total_composite/std": 5.023560152039863e-05, "sampling/importance_sampling_ratio/max": 1.3650829792022705, "sampling/importance_sampling_ratio/mean": 1.00238835811615, "sampling/importance_sampling_ratio/min": 0.8633265495300293, "sampling/sampling_logp_difference/max": 0.3112151622772217, "sampling/sampling_logp_difference/mean": 0.004402663093060255, "step": 2336 }, { "clip_ratio/high_max": 0.0036231884150765836, "clip_ratio/high_mean": 0.0036231884150765836, "clip_ratio/low_mean": 0.007231889001559466, "clip_ratio/low_min": 0.007231889001559466, "clip_ratio/region_mean": 0.01085507741663605, "completions/clipped_ratio": 0.0, "completions/max_length": 208.0, "completions/max_terminated_length": 208.0, "completions/mean_length": 207.0, "completions/mean_terminated_length": 207.0, "completions/min_length": 206.0, "completions/min_terminated_length": 206.0, "entropy": 0.09781584423035383, "epoch": 0.09386673093143752, "frac_reward_zero_std": 0.0, "grad_norm": 1.5094997882843018, "learning_rate": 2.9212121212121215e-06, "loss": 0.0008, "num_tokens": 5287016.0, "reward": 0.686035692691803, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978703260421753, "reward_meter_std": 3.329759420012124e-05, "reward_repeat_penalty_mean": 0.8020833730697632, "reward_repeat_penalty_std": 0.10853918641805649, "reward_std": 0.09283558279275894, "reward_total_composite_mean": 0.686035692691803, "reward_total_composite_std": 0.09283558279275894, "reward_total_mean": 0.686035692691803, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978703260421753, "rewards/meter/std": 3.329759420012124e-05, "rewards/repeat_penalty/mean": 0.8020833730697632, "rewards/repeat_penalty/std": 0.10853918641805649, "rewards/total_composite/mean": 0.686035692691803, "rewards/total_composite/std": 0.09283558279275894, "sampling/importance_sampling_ratio/max": 1.4435890913009644, "sampling/importance_sampling_ratio/mean": 1.0003283023834229, "sampling/importance_sampling_ratio/min": 0.34840553998947144, "sampling/sampling_logp_difference/max": 1.0543880462646484, "sampling/sampling_logp_difference/mean": 0.013423108495771885, "step": 2337 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.04713014606386423, "epoch": 0.09390689641322247, "frac_reward_zero_std": 0.0, "grad_norm": 0.4725949764251709, "learning_rate": 2.9181818181818183e-06, "loss": -0.0001, "num_tokens": 5288928.0, "reward": 0.9981162548065186, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981162548065186, "reward_meter_std": 2.446455619065091e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.446935832267627e-05, "reward_total_composite_mean": 0.9981162548065186, "reward_total_composite_std": 2.446455619065091e-05, "reward_total_mean": 0.9981162548065186, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981162548065186, "rewards/meter/std": 2.446455619065091e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981162548065186, "rewards/total_composite/std": 2.446455619065091e-05, "sampling/importance_sampling_ratio/max": 1.1632411479949951, "sampling/importance_sampling_ratio/mean": 0.9997783899307251, "sampling/importance_sampling_ratio/min": 0.42764657735824585, "sampling/sampling_logp_difference/max": 0.8494582176208496, "sampling/sampling_logp_difference/mean": 0.006431423127651215, "step": 2338 }, { "clip_ratio/high_max": 0.00484496122226119, "clip_ratio/high_mean": 0.00484496122226119, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/region_mean": 0.0067680381471291184, "completions/clipped_ratio": 0.0, "completions/max_length": 130.0, "completions/max_terminated_length": 130.0, "completions/mean_length": 129.25, "completions/mean_terminated_length": 129.25, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.040207551792263985, "epoch": 0.09394706189500743, "frac_reward_zero_std": 0.0, "grad_norm": 0.770327091217041, "learning_rate": 2.915151515151515e-06, "loss": 0.0017, "num_tokens": 5291330.0, "reward": 0.8553467988967896, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979046583175659, "reward_meter_std": 4.928474663756788e-05, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 4.226414966979064e-05, "reward_total_composite_mean": 0.8553467988967896, "reward_total_composite_std": 4.225334123475477e-05, "reward_total_mean": 0.8553467988967896, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979046583175659, "rewards/meter/std": 4.928474663756788e-05, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8553467988967896, "rewards/total_composite/std": 4.225334123475477e-05, "sampling/importance_sampling_ratio/max": 1.9889717102050781, "sampling/importance_sampling_ratio/mean": 0.9997851848602295, "sampling/importance_sampling_ratio/min": 0.23849685490131378, "sampling/sampling_logp_difference/max": 1.4333992004394531, "sampling/sampling_logp_difference/mean": 0.010605924762785435, "step": 2339 }, { "clip_ratio/high_max": 0.011448833451140672, "clip_ratio/high_mean": 0.011448833451140672, "clip_ratio/low_mean": 0.004446686594747007, "clip_ratio/low_min": 0.004446686594747007, "clip_ratio/region_mean": 0.01589552004588768, "completions/clipped_ratio": 0.0, "completions/max_length": 289.0, "completions/max_terminated_length": 289.0, "completions/mean_length": 262.875, "completions/mean_terminated_length": 262.875, "completions/min_length": 252.0, "completions/min_terminated_length": 252.0, "entropy": 0.18291304260492325, "epoch": 0.09398722737679238, "frac_reward_zero_std": 0.0, "grad_norm": 1.771281361579895, "learning_rate": 2.9121212121212124e-06, "loss": -0.0245, "num_tokens": 5295161.0, "reward": 0.6484043002128601, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8055555820465088, "reward_count_adherence_std": 0.05143444985151291, "reward_meter_mean": 0.9991946220397949, "reward_meter_std": 0.00014320346235763282, "reward_repeat_penalty_mean": 0.8057692050933838, "reward_repeat_penalty_std": 0.08968339115381241, "reward_std": 0.07946040481328964, "reward_total_composite_mean": 0.6484043002128601, "reward_total_composite_std": 0.07946042716503143, "reward_total_mean": 0.6484043002128601, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8055555820465088, "rewards/count_adherence/std": 0.05143444985151291, "rewards/meter/mean": 0.9991946220397949, "rewards/meter/std": 0.00014320346235763282, "rewards/repeat_penalty/mean": 0.8057692050933838, "rewards/repeat_penalty/std": 0.08968339115381241, "rewards/total_composite/mean": 0.6484043002128601, "rewards/total_composite/std": 0.07946042716503143, "sampling/importance_sampling_ratio/max": 1.6187031269073486, "sampling/importance_sampling_ratio/mean": 1.00568687915802, "sampling/importance_sampling_ratio/min": 0.3629041314125061, "sampling/sampling_logp_difference/max": 1.0136165618896484, "sampling/sampling_logp_difference/mean": 0.017933662980794907, "step": 2340 }, { "clip_ratio/high_max": 0.00551724131219089, "clip_ratio/high_mean": 0.00551724131219089, "clip_ratio/low_mean": 0.011894350522197783, "clip_ratio/low_min": 0.011894350522197783, "clip_ratio/region_mean": 0.017411591834388673, "completions/clipped_ratio": 0.0, "completions/max_length": 290.0, "completions/max_terminated_length": 290.0, "completions/mean_length": 265.625, "completions/mean_terminated_length": 265.625, "completions/min_length": 250.0, "completions/min_terminated_length": 250.0, "entropy": 0.20100159756839275, "epoch": 0.09402739285857734, "frac_reward_zero_std": 0.0, "grad_norm": 1.6412862539291382, "learning_rate": 2.9090909090909093e-06, "loss": -0.0113, "num_tokens": 5299382.0, "reward": 0.6096493005752563, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.737500011920929, "reward_count_adherence_std": 0.05175492912530899, "reward_meter_mean": 0.9990073442459106, "reward_meter_std": 0.0001355017302557826, "reward_repeat_penalty_mean": 0.8301282525062561, "reward_repeat_penalty_std": 0.06518138945102692, "reward_std": 0.03883464261889458, "reward_total_composite_mean": 0.6096493005752563, "reward_total_composite_std": 0.03883464261889458, "reward_total_mean": 0.6096493005752563, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.737500011920929, "rewards/count_adherence/std": 0.05175492912530899, "rewards/meter/mean": 0.9990073442459106, "rewards/meter/std": 0.0001355017302557826, "rewards/repeat_penalty/mean": 0.8301282525062561, "rewards/repeat_penalty/std": 0.06518138945102692, "rewards/total_composite/mean": 0.6096493005752563, "rewards/total_composite/std": 0.03883464261889458, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0052927732467651, "sampling/importance_sampling_ratio/min": 0.30119454860687256, "sampling/sampling_logp_difference/max": 2.072540044784546, "sampling/sampling_logp_difference/mean": 0.022991513833403587, "step": 2341 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.003808505309280008, "epoch": 0.09406755834036229, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.906060606060606e-06, "loss": 0.0, "num_tokens": 5301206.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0067293643951416, "sampling/importance_sampling_ratio/mean": 1.000400185585022, "sampling/importance_sampling_ratio/min": 0.9993113875389099, "sampling/sampling_logp_difference/max": 0.0067067258059978485, "sampling/sampling_logp_difference/mean": 0.00041229455382563174, "step": 2342 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0037699798413086683, "epoch": 0.09410772382214724, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.903030303030303e-06, "loss": 0.0, "num_tokens": 5303262.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0098844766616821, "sampling/importance_sampling_ratio/mean": 1.0003089904785156, "sampling/importance_sampling_ratio/min": 0.9851569533348083, "sampling/sampling_logp_difference/max": 0.01495426706969738, "sampling/sampling_logp_difference/mean": 0.00038630847120657563, "step": 2343 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0038317264115903527, "epoch": 0.0941478893039322, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.9e-06, "loss": 0.0, "num_tokens": 5305230.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0104482173919678, "sampling/importance_sampling_ratio/mean": 1.000275731086731, "sampling/importance_sampling_ratio/min": 0.989374041557312, "sampling/sampling_logp_difference/max": 0.01068283524364233, "sampling/sampling_logp_difference/mean": 0.00038092854083515704, "step": 2344 }, { "clip_ratio/high_max": 0.029499810189008713, "clip_ratio/high_mean": 0.029499810189008713, "clip_ratio/low_mean": 0.023096519173122942, "clip_ratio/low_min": 0.023096519173122942, "clip_ratio/region_mean": 0.052596329362131655, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 64.375, "completions/mean_terminated_length": 64.375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.3568100146949291, "epoch": 0.09418805478571715, "frac_reward_zero_std": 0.0, "grad_norm": 9.732110023498535, "learning_rate": 2.896969696969697e-06, "loss": 0.0324, "num_tokens": 5306953.0, "reward": 0.7001276016235352, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7194586992263794, "reward_meter_std": 0.26375213265419006, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.26491162180900574, "reward_total_composite_mean": 0.7001276016235352, "reward_total_composite_std": 0.2649116516113281, "reward_total_mean": 0.7001276016235352, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7194586992263794, "rewards/meter/std": 0.26375213265419006, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.7001276016235352, "rewards/total_composite/std": 0.2649116516113281, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0110507011413574, "sampling/importance_sampling_ratio/min": 0.12938177585601807, "sampling/sampling_logp_difference/max": 2.044987678527832, "sampling/sampling_logp_difference/mean": 0.06900904327630997, "step": 2345 }, { "clip_ratio/high_max": 0.02505582571029663, "clip_ratio/high_mean": 0.02505582571029663, "clip_ratio/low_mean": 0.014375669648870826, "clip_ratio/low_min": 0.014375669648870826, "clip_ratio/region_mean": 0.03943149535916746, "completions/clipped_ratio": 0.0, "completions/max_length": 248.0, "completions/max_terminated_length": 248.0, "completions/mean_length": 237.875, "completions/mean_terminated_length": 237.875, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.4049236960709095, "epoch": 0.0942282202675021, "frac_reward_zero_std": 0.0, "grad_norm": 3.087636947631836, "learning_rate": 2.893939393939394e-06, "loss": -0.0015, "num_tokens": 5310264.0, "reward": 0.9405449032783508, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981396794319153, "reward_meter_std": 0.0003048716171178967, "reward_repeat_penalty_mean": 0.942307710647583, "reward_repeat_penalty_std": 0.07962294667959213, "reward_std": 0.07935541868209839, "reward_total_composite_mean": 0.9405449032783508, "reward_total_composite_std": 0.07935542613267899, "reward_total_mean": 0.9405449032783508, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981396794319153, "rewards/meter/std": 0.0003048716171178967, "rewards/repeat_penalty/mean": 0.942307710647583, "rewards/repeat_penalty/std": 0.07962294667959213, "rewards/total_composite/mean": 0.9405449032783508, "rewards/total_composite/std": 0.07935542613267899, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0058072805404663, "sampling/importance_sampling_ratio/min": 0.003953665494918823, "sampling/sampling_logp_difference/max": 5.533112049102783, "sampling/sampling_logp_difference/mean": 0.0508662573993206, "step": 2346 }, { "clip_ratio/high_max": 0.031063538044691086, "clip_ratio/high_mean": 0.031063538044691086, "clip_ratio/low_mean": 0.017315812641754746, "clip_ratio/low_min": 0.017315812641754746, "clip_ratio/region_mean": 0.04837935068644583, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 100.5, "completions/mean_terminated_length": 100.5, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.4347502291202545, "epoch": 0.09426838574928706, "frac_reward_zero_std": 0.0, "grad_norm": 3.2611429691314697, "learning_rate": 2.8909090909090907e-06, "loss": -0.0006, "num_tokens": 5312388.0, "reward": 0.9978408217430115, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978408217430115, "reward_meter_std": 0.0007257388206198812, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007257444667629898, "reward_total_composite_mean": 0.9978408217430115, "reward_total_composite_std": 0.0007257388206198812, "reward_total_mean": 0.9978408217430115, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978408217430115, "rewards/meter/std": 0.0007257388206198812, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978408217430115, "rewards/total_composite/std": 0.0007257388206198812, "sampling/importance_sampling_ratio/max": 1.796935796737671, "sampling/importance_sampling_ratio/mean": 1.0071477890014648, "sampling/importance_sampling_ratio/min": 0.09898723661899567, "sampling/sampling_logp_difference/max": 2.3127644062042236, "sampling/sampling_logp_difference/mean": 0.056275609880685806, "step": 2347 }, { "clip_ratio/high_max": 0.003007704159244895, "clip_ratio/high_mean": 0.003007704159244895, "clip_ratio/low_mean": 0.00988566328305751, "clip_ratio/low_min": 0.00988566328305751, "clip_ratio/region_mean": 0.012893367442302406, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 165.0, "completions/mean_terminated_length": 165.0, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.10382581874728203, "epoch": 0.09430855123107201, "frac_reward_zero_std": 0.0, "grad_norm": 2.218722105026245, "learning_rate": 2.8878787878787884e-06, "loss": -0.0016, "num_tokens": 5315212.0, "reward": 0.9294023513793945, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987882375717163, "reward_meter_std": 0.0007174185593612492, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.05695149302482605, "reward_total_composite_mean": 0.9294023513793945, "reward_total_composite_std": 0.05695149675011635, "reward_total_mean": 0.9294023513793945, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987882375717163, "rewards/meter/std": 0.0007174185593612492, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9294023513793945, "rewards/total_composite/std": 0.05695149675011635, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0003210306167603, "sampling/importance_sampling_ratio/min": 0.28630349040031433, "sampling/sampling_logp_difference/max": 1.2507028579711914, "sampling/sampling_logp_difference/mean": 0.018433764576911926, "step": 2348 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.013554216362535954, "clip_ratio/low_min": 0.013554216362535954, "clip_ratio/region_mean": 0.013554216362535954, "completions/clipped_ratio": 0.0, "completions/max_length": 84.0, "completions/max_terminated_length": 84.0, "completions/mean_length": 83.125, "completions/mean_terminated_length": 83.125, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.04944844264537096, "epoch": 0.09434871671285697, "frac_reward_zero_std": 0.0, "grad_norm": 2.1028835773468018, "learning_rate": 2.884848484848485e-06, "loss": -0.0002, "num_tokens": 5317189.0, "reward": 0.9964075684547424, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9964075684547424, "reward_meter_std": 0.0002765149693004787, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002765233220998198, "reward_total_composite_mean": 0.9964075684547424, "reward_total_composite_std": 0.0002765149693004787, "reward_total_mean": 0.9964075684547424, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9964075684547424, "rewards/meter/std": 0.0002765149693004787, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964075684547424, "rewards/total_composite/std": 0.0002765149693004787, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.00690495967865, "sampling/importance_sampling_ratio/min": 0.3692053556442261, "sampling/sampling_logp_difference/max": 0.9964022636413574, "sampling/sampling_logp_difference/mean": 0.014425117522478104, "step": 2349 }, { "clip_ratio/high_max": 0.03450171835720539, "clip_ratio/high_mean": 0.03450171835720539, "clip_ratio/low_mean": 0.004700823919847608, "clip_ratio/low_min": 0.004700823919847608, "clip_ratio/region_mean": 0.039202542277053, "completions/clipped_ratio": 0.0, "completions/max_length": 203.0, "completions/max_terminated_length": 203.0, "completions/mean_length": 193.25, "completions/mean_terminated_length": 193.25, "completions/min_length": 185.0, "completions/min_terminated_length": 185.0, "entropy": 0.40191996842622757, "epoch": 0.09438888219464192, "frac_reward_zero_std": 0.0, "grad_norm": 2.5682480335235596, "learning_rate": 2.8818181818181824e-06, "loss": -0.0123, "num_tokens": 5320279.0, "reward": 0.9640417098999023, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980719089508057, "reward_meter_std": 0.00038988571031950414, "reward_repeat_penalty_mean": 0.9659091234207153, "reward_repeat_penalty_std": 0.06763853132724762, "reward_std": 0.06744074076414108, "reward_total_composite_mean": 0.9640417098999023, "reward_total_composite_std": 0.06744071841239929, "reward_total_mean": 0.9640417098999023, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980719089508057, "rewards/meter/std": 0.00038988571031950414, "rewards/repeat_penalty/mean": 0.9659091234207153, "rewards/repeat_penalty/std": 0.06763853132724762, "rewards/total_composite/mean": 0.9640417098999023, "rewards/total_composite/std": 0.06744071841239929, "sampling/importance_sampling_ratio/max": 1.9953590631484985, "sampling/importance_sampling_ratio/mean": 1.0081247091293335, "sampling/importance_sampling_ratio/min": 0.2854640483856201, "sampling/sampling_logp_difference/max": 1.2536392211914062, "sampling/sampling_logp_difference/mean": 0.05063340812921524, "step": 2350 }, { "epoch": 0.09438888219464192, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 349.3076923076923, "eval_completions/max_terminated_length": 349.3076923076923, "eval_completions/mean_length": 194.6346153846154, "eval_completions/mean_terminated_length": 194.6346153846154, "eval_completions/min_length": 61.30769230769231, "eval_completions/min_terminated_length": 61.30769230769231, "eval_entropy": 0.21742234894862542, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 5320279.0, "eval_reward": 0.6291824304140531, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9230444156206571, "eval_reward_count_adherence_std": 0.10953088907095102, "eval_reward_meter_mean": 0.7545687923064599, "eval_reward_meter_std": 0.3586408014480884, "eval_reward_repeat_penalty_mean": 0.885237611257113, "eval_reward_repeat_penalty_std": 0.1111061366704794, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6291824304140531, "eval_reward_total_composite_std": 0.33865179121494293, "eval_reward_total_mean": 0.6291824304140531, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9230444156206571, "eval_rewards/count_adherence/std": 0.10953088907095102, "eval_rewards/meter/mean": 0.7545687923064599, "eval_rewards/meter/std": 0.3586408014480884, "eval_rewards/repeat_penalty/mean": 0.885237611257113, "eval_rewards/repeat_penalty/std": 0.1111061366704794, "eval_rewards/total_composite/mean": 0.6291824304140531, "eval_rewards/total_composite/std": 0.33865179121494293, "eval_runtime": 67.6666, "eval_samples_per_second": 1.537, "eval_sampling/importance_sampling_ratio/max": 1.4322827320832472, "eval_sampling/importance_sampling_ratio/mean": 1.0047486470295832, "eval_sampling/importance_sampling_ratio/min": 0.3466451099285713, "eval_sampling/sampling_logp_difference/max": 1.0696188119741588, "eval_sampling/sampling_logp_difference/mean": 0.021095395159835998, "eval_steps_per_second": 0.192, "step": 2350 }, { "clip_ratio/high_max": 0.01327340246643871, "clip_ratio/high_mean": 0.01327340246643871, "clip_ratio/low_mean": 0.008248508209362626, "clip_ratio/low_min": 0.008248508209362626, "clip_ratio/region_mean": 0.021521910675801337, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 75.5, "completions/mean_terminated_length": 75.5, "completions/min_length": 73.0, "completions/min_terminated_length": 73.0, "entropy": 0.20778764225542545, "epoch": 0.09442904767642687, "frac_reward_zero_std": 0.0, "grad_norm": 4.469809532165527, "learning_rate": 2.8787878787878793e-06, "loss": 0.0052, "num_tokens": 5322067.0, "reward": 0.998935878276825, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998935878276825, "reward_meter_std": 0.0006324206478893757, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006324192509055138, "reward_total_composite_mean": 0.998935878276825, "reward_total_composite_std": 0.0006324206478893757, "reward_total_mean": 0.998935878276825, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998935878276825, "rewards/meter/std": 0.0006324206478893757, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998935878276825, "rewards/total_composite/std": 0.0006324206478893757, "sampling/importance_sampling_ratio/max": 1.4723999500274658, "sampling/importance_sampling_ratio/mean": 1.0036641359329224, "sampling/importance_sampling_ratio/min": 0.33441799879074097, "sampling/sampling_logp_difference/max": 1.0953636169433594, "sampling/sampling_logp_difference/mean": 0.030197838321328163, "step": 2351 }, { "clip_ratio/high_max": 0.004545454401522875, "clip_ratio/high_mean": 0.004545454401522875, "clip_ratio/low_mean": 0.0022727272007614374, "clip_ratio/low_min": 0.0022727272007614374, "clip_ratio/region_mean": 0.006818181602284312, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 55.0, "completions/mean_terminated_length": 55.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.018420133623294532, "epoch": 0.09446921315821183, "frac_reward_zero_std": 0.0, "grad_norm": 2.229356288909912, "learning_rate": 2.875757575757576e-06, "loss": -0.0028, "num_tokens": 5323867.0, "reward": 0.9961405992507935, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961405992507935, "reward_meter_std": 0.0004942560917697847, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004942500381730497, "reward_total_composite_mean": 0.9961405992507935, "reward_total_composite_std": 0.0004942560917697847, "reward_total_mean": 0.9961405992507935, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961405992507935, "rewards/meter/std": 0.0004942560917697847, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9961405992507935, "rewards/total_composite/std": 0.0004942560917697847, "sampling/importance_sampling_ratio/max": 1.1960804462432861, "sampling/importance_sampling_ratio/mean": 0.9985120892524719, "sampling/importance_sampling_ratio/min": 0.2581455409526825, "sampling/sampling_logp_difference/max": 1.3542317152023315, "sampling/sampling_logp_difference/mean": 0.006679588928818703, "step": 2352 }, { "clip_ratio/high_max": 0.007577877026051283, "clip_ratio/high_mean": 0.007577877026051283, "clip_ratio/low_mean": 0.01009113050531596, "clip_ratio/low_min": 0.01009113050531596, "clip_ratio/region_mean": 0.017669007531367242, "completions/clipped_ratio": 0.0, "completions/max_length": 455.0, "completions/max_terminated_length": 455.0, "completions/mean_length": 421.375, "completions/mean_terminated_length": 421.375, "completions/min_length": 398.0, "completions/min_terminated_length": 398.0, "entropy": 0.19907046295702457, "epoch": 0.09450937863999678, "frac_reward_zero_std": 0.0, "grad_norm": 1.3235185146331787, "learning_rate": 2.872727272727273e-06, "loss": -0.0155, "num_tokens": 5329206.0, "reward": 0.5467392206192017, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6470588445663452, "reward_count_adherence_std": 0.03144249692559242, "reward_meter_mean": 0.9977288246154785, "reward_meter_std": 0.0008687536465004086, "reward_repeat_penalty_mean": 0.8467603325843811, "reward_repeat_penalty_std": 0.08341042697429657, "reward_std": 0.061857227236032486, "reward_total_composite_mean": 0.5467392206192017, "reward_total_composite_std": 0.06185723468661308, "reward_total_mean": 0.5467392206192017, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6470588445663452, "rewards/count_adherence/std": 0.03144249692559242, "rewards/meter/mean": 0.9977288246154785, "rewards/meter/std": 0.0008687536465004086, "rewards/repeat_penalty/mean": 0.8467603325843811, "rewards/repeat_penalty/std": 0.08341042697429657, "rewards/total_composite/mean": 0.5467392206192017, "rewards/total_composite/std": 0.06185723468661308, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0037055015563965, "sampling/importance_sampling_ratio/min": 0.1462891846895218, "sampling/sampling_logp_difference/max": 1.9221699237823486, "sampling/sampling_logp_difference/mean": 0.02322297915816307, "step": 2353 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0035217964032199234, "epoch": 0.09454954412178174, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.86969696969697e-06, "loss": 0.0, "num_tokens": 5330886.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0068386793136597, "sampling/importance_sampling_ratio/mean": 1.0003925561904907, "sampling/importance_sampling_ratio/min": 1.0, "sampling/sampling_logp_difference/max": 0.0068152910098433495, "sampling/sampling_logp_difference/mean": 0.0003920203307643533, "step": 2354 }, { "clip_ratio/high_max": 0.017101168166846037, "clip_ratio/high_mean": 0.017101168166846037, "clip_ratio/low_mean": 0.008841085364110768, "clip_ratio/low_min": 0.008841085364110768, "clip_ratio/region_mean": 0.025942253530956805, "completions/clipped_ratio": 0.0, "completions/max_length": 171.0, "completions/max_terminated_length": 171.0, "completions/mean_length": 168.75, "completions/mean_terminated_length": 168.75, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.1752516869455576, "epoch": 0.09458970960356669, "frac_reward_zero_std": 0.0, "grad_norm": 2.712887763977051, "learning_rate": 2.866666666666667e-06, "loss": 0.0071, "num_tokens": 5333484.0, "reward": 0.8432746529579163, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9781259298324585, "reward_meter_std": 0.03459608182311058, "reward_repeat_penalty_mean": 0.8611111640930176, "reward_repeat_penalty_std": 0.07856741547584534, "reward_std": 0.09240438789129257, "reward_total_composite_mean": 0.8432746529579163, "reward_total_composite_std": 0.09240440279245377, "reward_total_mean": 0.8432746529579163, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9781259298324585, "rewards/meter/std": 0.03459608182311058, "rewards/repeat_penalty/mean": 0.8611111640930176, "rewards/repeat_penalty/std": 0.07856741547584534, "rewards/total_composite/mean": 0.8432746529579163, "rewards/total_composite/std": 0.09240440279245377, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0031359195709229, "sampling/importance_sampling_ratio/min": 0.2791994512081146, "sampling/sampling_logp_difference/max": 1.2758288383483887, "sampling/sampling_logp_difference/mean": 0.027000876143574715, "step": 2355 }, { "clip_ratio/high_max": 0.01955222897231579, "clip_ratio/high_mean": 0.01955222897231579, "clip_ratio/low_mean": 0.030935976887121797, "clip_ratio/low_min": 0.030935976887121797, "clip_ratio/region_mean": 0.050488205859437585, "completions/clipped_ratio": 0.0, "completions/max_length": 49.0, "completions/max_terminated_length": 49.0, "completions/mean_length": 44.375, "completions/mean_terminated_length": 44.375, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.32474897243082523, "epoch": 0.09462987508535164, "frac_reward_zero_std": 0.0, "grad_norm": 12.347071647644043, "learning_rate": 2.863636363636364e-06, "loss": 0.034, "num_tokens": 5335199.0, "reward": 0.9444069266319275, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9444069266319275, "reward_meter_std": 0.008035099133849144, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00803509633988142, "reward_total_composite_mean": 0.9444069266319275, "reward_total_composite_std": 0.008035099133849144, "reward_total_mean": 0.9444069266319275, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9444069266319275, "rewards/meter/std": 0.008035099133849144, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9444069266319275, "rewards/total_composite/std": 0.008035099133849144, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0038548707962036, "sampling/importance_sampling_ratio/min": 0.3516010642051697, "sampling/sampling_logp_difference/max": 2.617283821105957, "sampling/sampling_logp_difference/mean": 0.07534459978342056, "step": 2356 }, { "clip_ratio/high_max": 0.009516005055047572, "clip_ratio/high_mean": 0.009516005055047572, "clip_ratio/low_mean": 0.00634939968585968, "clip_ratio/low_min": 0.00634939968585968, "clip_ratio/region_mean": 0.015865404740907252, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 78.625, "completions/mean_terminated_length": 78.625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.12107675615698099, "epoch": 0.0946700405671366, "frac_reward_zero_std": 0.0, "grad_norm": 2.1621651649475098, "learning_rate": 2.860606060606061e-06, "loss": -0.0003, "num_tokens": 5337204.0, "reward": 0.9971969127655029, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971969127655029, "reward_meter_std": 0.0005990342469885945, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005990189965814352, "reward_total_composite_mean": 0.9971969127655029, "reward_total_composite_std": 0.0005990342469885945, "reward_total_mean": 0.9971969127655029, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971969127655029, "rewards/meter/std": 0.0005990342469885945, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971969127655029, "rewards/total_composite/std": 0.0005990342469885945, "sampling/importance_sampling_ratio/max": 1.4412025213241577, "sampling/importance_sampling_ratio/mean": 0.9987841844558716, "sampling/importance_sampling_ratio/min": 0.2547987103462219, "sampling/sampling_logp_difference/max": 1.367281436920166, "sampling/sampling_logp_difference/mean": 0.021038906648755074, "step": 2357 }, { "clip_ratio/high_max": 0.008355856058187783, "clip_ratio/high_mean": 0.008355856058187783, "clip_ratio/low_mean": 0.02065259451046586, "clip_ratio/low_min": 0.02065259451046586, "clip_ratio/region_mean": 0.029008450568653643, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 73.625, "completions/mean_terminated_length": 73.625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.25467516854405403, "epoch": 0.09471020604892155, "frac_reward_zero_std": 0.0, "grad_norm": 3.5978167057037354, "learning_rate": 2.857575757575758e-06, "loss": -0.0096, "num_tokens": 5339049.0, "reward": 0.9990395307540894, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990395307540894, "reward_meter_std": 0.00045778730418533087, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004578051157295704, "reward_total_composite_mean": 0.9990395307540894, "reward_total_composite_std": 0.00045778730418533087, "reward_total_mean": 0.9990395307540894, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990395307540894, "rewards/meter/std": 0.00045778730418533087, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990395307540894, "rewards/total_composite/std": 0.00045778730418533087, "sampling/importance_sampling_ratio/max": 1.805698275566101, "sampling/importance_sampling_ratio/mean": 0.999204158782959, "sampling/importance_sampling_ratio/min": 0.2094704955816269, "sampling/sampling_logp_difference/max": 1.5631723403930664, "sampling/sampling_logp_difference/mean": 0.0385035015642643, "step": 2358 }, { "clip_ratio/high_max": 0.016904115793295205, "clip_ratio/high_mean": 0.016904115793295205, "clip_ratio/low_mean": 0.007326300255954266, "clip_ratio/low_min": 0.007326300255954266, "clip_ratio/region_mean": 0.02423041604924947, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2618895582854748, "epoch": 0.0947503715307065, "frac_reward_zero_std": 0.0, "grad_norm": 6.299208164215088, "learning_rate": 2.8545454545454548e-06, "loss": 0.015, "num_tokens": 5340832.0, "reward": 0.9785959720611572, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9785959720611572, "reward_meter_std": 0.018975911661982536, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.018975909799337387, "reward_total_composite_mean": 0.9785959720611572, "reward_total_composite_std": 0.018975911661982536, "reward_total_mean": 0.9785959720611572, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9785959720611572, "rewards/meter/std": 0.018975911661982536, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9785959720611572, "rewards/total_composite/std": 0.018975911661982536, "sampling/importance_sampling_ratio/max": 1.6261426210403442, "sampling/importance_sampling_ratio/mean": 1.0064629316329956, "sampling/importance_sampling_ratio/min": 0.27349260449409485, "sampling/sampling_logp_difference/max": 1.296480655670166, "sampling/sampling_logp_difference/mean": 0.0360257662832737, "step": 2359 }, { "clip_ratio/high_max": 0.03938204434234649, "clip_ratio/high_mean": 0.03938204434234649, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/region_mean": 0.04116775863803923, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.30276480689644814, "epoch": 0.09479053701249146, "frac_reward_zero_std": 0.0, "grad_norm": 5.956533432006836, "learning_rate": 2.8515151515151516e-06, "loss": 0.0207, "num_tokens": 5342577.0, "reward": 0.9699335098266602, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9699335098266602, "reward_meter_std": 0.04688005894422531, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04688006266951561, "reward_total_composite_mean": 0.9699335098266602, "reward_total_composite_std": 0.04688005894422531, "reward_total_mean": 0.9699335098266602, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9699335098266602, "rewards/meter/std": 0.04688005894422531, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9699335098266602, "rewards/total_composite/std": 0.04688005894422531, "sampling/importance_sampling_ratio/max": 1.829169750213623, "sampling/importance_sampling_ratio/mean": 1.0127477645874023, "sampling/importance_sampling_ratio/min": 0.36333972215652466, "sampling/sampling_logp_difference/max": 1.0124170780181885, "sampling/sampling_logp_difference/mean": 0.03932439908385277, "step": 2360 }, { "clip_ratio/high_max": 0.00470314035192132, "clip_ratio/high_mean": 0.00470314035192132, "clip_ratio/low_mean": 0.009289931389503181, "clip_ratio/low_min": 0.009289931389503181, "clip_ratio/region_mean": 0.013993071741424501, "completions/clipped_ratio": 0.0, "completions/max_length": 219.0, "completions/max_terminated_length": 219.0, "completions/mean_length": 214.5, "completions/mean_terminated_length": 214.5, "completions/min_length": 210.0, "completions/min_terminated_length": 210.0, "entropy": 0.19607393071055412, "epoch": 0.09483070249427641, "frac_reward_zero_std": 0.0, "grad_norm": 1.9891157150268555, "learning_rate": 2.848484848484849e-06, "loss": 0.0078, "num_tokens": 5345741.0, "reward": 0.8401910066604614, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999156653881073, "reward_meter_std": 0.00014262554759625345, "reward_repeat_penalty_mean": 0.8409091234207153, "reward_repeat_penalty_std": 0.10590588301420212, "reward_std": 0.10572723299264908, "reward_total_composite_mean": 0.8401910066604614, "reward_total_composite_std": 0.10572723299264908, "reward_total_mean": 0.8401910066604614, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999156653881073, "rewards/meter/std": 0.00014262554759625345, "rewards/repeat_penalty/mean": 0.8409091234207153, "rewards/repeat_penalty/std": 0.10590588301420212, "rewards/total_composite/mean": 0.8401910066604614, "rewards/total_composite/std": 0.10572723299264908, "sampling/importance_sampling_ratio/max": 1.6807578802108765, "sampling/importance_sampling_ratio/mean": 1.002423882484436, "sampling/importance_sampling_ratio/min": 0.29979807138442993, "sampling/sampling_logp_difference/max": 1.204646110534668, "sampling/sampling_logp_difference/mean": 0.021222621202468872, "step": 2361 }, { "clip_ratio/high_max": 0.006097560748457909, "clip_ratio/high_mean": 0.006097560748457909, "clip_ratio/low_mean": 0.0032051282469183207, "clip_ratio/low_min": 0.0032051282469183207, "clip_ratio/region_mean": 0.00930268899537623, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 40.375, "completions/mean_terminated_length": 40.375, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.08906339993700385, "epoch": 0.09487086797606137, "frac_reward_zero_std": 0.0, "grad_norm": 3.921265125274658, "learning_rate": 2.8454545454545457e-06, "loss": -0.0082, "num_tokens": 5347168.0, "reward": 0.9963768124580383, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963768124580383, "reward_meter_std": 0.0008168795611709356, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008168675121851265, "reward_total_composite_mean": 0.9963768124580383, "reward_total_composite_std": 0.0008168795611709356, "reward_total_mean": 0.9963768124580383, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963768124580383, "rewards/meter/std": 0.0008168795611709356, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963768124580383, "rewards/total_composite/std": 0.0008168795611709356, "sampling/importance_sampling_ratio/max": 1.4037843942642212, "sampling/importance_sampling_ratio/mean": 0.9972823858261108, "sampling/importance_sampling_ratio/min": 0.5636902451515198, "sampling/sampling_logp_difference/max": 0.573250412940979, "sampling/sampling_logp_difference/mean": 0.018412932753562927, "step": 2362 }, { "clip_ratio/high_max": 0.021532694692723453, "clip_ratio/high_mean": 0.021532694692723453, "clip_ratio/low_mean": 0.00890342053025961, "clip_ratio/low_min": 0.00890342053025961, "clip_ratio/region_mean": 0.030436115222983062, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 74.375, "completions/mean_terminated_length": 74.375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.2678355220705271, "epoch": 0.09491103345784632, "frac_reward_zero_std": 0.0, "grad_norm": 4.108429908752441, "learning_rate": 2.8424242424242425e-06, "loss": -0.0237, "num_tokens": 5348939.0, "reward": 0.9990479946136475, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990479946136475, "reward_meter_std": 0.0008222111500799656, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008222072501666844, "reward_total_composite_mean": 0.9990479946136475, "reward_total_composite_std": 0.0008222111500799656, "reward_total_mean": 0.9990479946136475, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990479946136475, "rewards/meter/std": 0.0008222111500799656, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990479946136475, "rewards/total_composite/std": 0.0008222111500799656, "sampling/importance_sampling_ratio/max": 1.825773000717163, "sampling/importance_sampling_ratio/mean": 0.9991539120674133, "sampling/importance_sampling_ratio/min": 0.3791051506996155, "sampling/sampling_logp_difference/max": 0.9699416160583496, "sampling/sampling_logp_difference/mean": 0.03708229959011078, "step": 2363 }, { "clip_ratio/high_max": 0.01258992834482342, "clip_ratio/high_mean": 0.01258992834482342, "clip_ratio/low_mean": 0.0054152331431396306, "clip_ratio/low_min": 0.0054152331431396306, "clip_ratio/region_mean": 0.01800516148796305, "completions/clipped_ratio": 0.0, "completions/max_length": 139.0, "completions/max_terminated_length": 139.0, "completions/mean_length": 138.875, "completions/mean_terminated_length": 138.875, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "entropy": 0.07086402643471956, "epoch": 0.09495119893963128, "frac_reward_zero_std": 0.0, "grad_norm": 1.7912240028381348, "learning_rate": 2.83939393939394e-06, "loss": 0.0027, "num_tokens": 5351730.0, "reward": 0.8302946090698242, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963522553443909, "reward_meter_std": 0.0003084685595240444, "reward_repeat_penalty_mean": 0.8333333730697632, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.05919152498245239, "reward_total_composite_mean": 0.8302946090698242, "reward_total_composite_std": 0.0591915100812912, "reward_total_mean": 0.8302946090698242, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963522553443909, "rewards/meter/std": 0.0003084685595240444, "rewards/repeat_penalty/mean": 0.8333333730697632, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.8302946090698242, "rewards/total_composite/std": 0.0591915100812912, "sampling/importance_sampling_ratio/max": 1.877577304840088, "sampling/importance_sampling_ratio/mean": 0.9980447292327881, "sampling/importance_sampling_ratio/min": 0.2732773423194885, "sampling/sampling_logp_difference/max": 1.2972681522369385, "sampling/sampling_logp_difference/mean": 0.015195811167359352, "step": 2364 }, { "clip_ratio/high_max": 0.01530504459515214, "clip_ratio/high_mean": 0.01530504459515214, "clip_ratio/low_mean": 0.010369744850322604, "clip_ratio/low_min": 0.010369744850322604, "clip_ratio/region_mean": 0.025674789445474744, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 73.25, "completions/mean_terminated_length": 73.25, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.20148388668894768, "epoch": 0.09499136442141623, "frac_reward_zero_std": 0.0, "grad_norm": 0.9692533612251282, "learning_rate": 2.8363636363636366e-06, "loss": -0.0049, "num_tokens": 5353652.0, "reward": 0.9992170333862305, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992170333862305, "reward_meter_std": 0.00020559101540129632, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020559433323796839, "reward_total_composite_mean": 0.9992170333862305, "reward_total_composite_std": 0.00020559101540129632, "reward_total_mean": 0.9992170333862305, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992170333862305, "rewards/meter/std": 0.00020559101540129632, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992170333862305, "rewards/total_composite/std": 0.00020559101540129632, "sampling/importance_sampling_ratio/max": 1.8762344121932983, "sampling/importance_sampling_ratio/mean": 1.0054048299789429, "sampling/importance_sampling_ratio/min": 0.20411363244056702, "sampling/sampling_logp_difference/max": 1.589078426361084, "sampling/sampling_logp_difference/mean": 0.023841191083192825, "step": 2365 }, { "clip_ratio/high_max": 0.03128110943362117, "clip_ratio/high_mean": 0.03128110943362117, "clip_ratio/low_mean": 0.01210981048643589, "clip_ratio/low_min": 0.01210981048643589, "clip_ratio/region_mean": 0.04339091992005706, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 132.125, "completions/mean_terminated_length": 132.125, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.3485673386603594, "epoch": 0.09503152990320118, "frac_reward_zero_std": 0.0, "grad_norm": 4.017276763916016, "learning_rate": 2.8333333333333335e-06, "loss": 0.0089, "num_tokens": 5356221.0, "reward": 0.9623122215270996, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979519844055176, "reward_meter_std": 0.0010567301651462913, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06602419167757034, "reward_total_composite_mean": 0.9623122215270996, "reward_total_composite_std": 0.06602418422698975, "reward_total_mean": 0.9623122215270996, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979519844055176, "rewards/meter/std": 0.0010567301651462913, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9623122215270996, "rewards/total_composite/std": 0.06602418422698975, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0044370889663696, "sampling/importance_sampling_ratio/min": 3.95094248233363e-05, "sampling/sampling_logp_difference/max": 10.138971328735352, "sampling/sampling_logp_difference/mean": 0.05144790932536125, "step": 2366 }, { "clip_ratio/high_max": 0.009202249813824892, "clip_ratio/high_mean": 0.009202249813824892, "clip_ratio/low_mean": 0.007995150808710605, "clip_ratio/low_min": 0.007995150808710605, "clip_ratio/region_mean": 0.017197400622535497, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 154.25, "completions/mean_terminated_length": 154.25, "completions/min_length": 147.0, "completions/min_terminated_length": 147.0, "entropy": 0.16689980775117874, "epoch": 0.09507169538498614, "frac_reward_zero_std": 0.0, "grad_norm": 1.9662439823150635, "learning_rate": 2.8303030303030303e-06, "loss": 0.0164, "num_tokens": 5359095.0, "reward": 0.9088918566703796, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9776645302772522, "reward_meter_std": 0.028479592874646187, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.04527008906006813, "reward_total_composite_mean": 0.9088918566703796, "reward_total_composite_std": 0.04527009651064873, "reward_total_mean": 0.9088918566703796, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9776645302772522, "rewards/meter/std": 0.028479592874646187, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9088918566703796, "rewards/total_composite/std": 0.04527009651064873, "sampling/importance_sampling_ratio/max": 1.7584105730056763, "sampling/importance_sampling_ratio/mean": 1.0010682344436646, "sampling/importance_sampling_ratio/min": 0.19941700994968414, "sampling/sampling_logp_difference/max": 1.6123571395874023, "sampling/sampling_logp_difference/mean": 0.023603463545441628, "step": 2367 }, { "clip_ratio/high_max": 0.02565775695256889, "clip_ratio/high_mean": 0.02565775695256889, "clip_ratio/low_mean": 0.01125139684882015, "clip_ratio/low_min": 0.01125139684882015, "clip_ratio/region_mean": 0.03690915380138904, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.75, "completions/mean_terminated_length": 67.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2899752501398325, "epoch": 0.09511186086677109, "frac_reward_zero_std": 0.0, "grad_norm": 3.341148853302002, "learning_rate": 2.8272727272727275e-06, "loss": -0.005, "num_tokens": 5360933.0, "reward": 0.998802125453949, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998802125453949, "reward_meter_std": 0.00023434955801349133, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00023435504408553243, "reward_total_composite_mean": 0.998802125453949, "reward_total_composite_std": 0.00023434955801349133, "reward_total_mean": 0.998802125453949, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998802125453949, "rewards/meter/std": 0.00023434955801349133, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998802125453949, "rewards/total_composite/std": 0.00023434955801349133, "sampling/importance_sampling_ratio/max": 1.526517629623413, "sampling/importance_sampling_ratio/mean": 1.0024096965789795, "sampling/importance_sampling_ratio/min": 0.40534451603889465, "sampling/sampling_logp_difference/max": 0.9030179977416992, "sampling/sampling_logp_difference/mean": 0.036128342151641846, "step": 2368 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.005597014795057476, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.09162188414484262, "epoch": 0.09515202634855605, "frac_reward_zero_std": 0.0, "grad_norm": 1.1244981288909912, "learning_rate": 2.8242424242424244e-06, "loss": 0.0001, "num_tokens": 5362693.0, "reward": 0.9980802536010742, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980802536010742, "reward_meter_std": 5.821814920636825e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.822332968818955e-05, "reward_total_composite_mean": 0.9980802536010742, "reward_total_composite_std": 5.821814920636825e-05, "reward_total_mean": 0.9980802536010742, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980802536010742, "rewards/meter/std": 5.821814920636825e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980802536010742, "rewards/total_composite/std": 5.821814920636825e-05, "sampling/importance_sampling_ratio/max": 1.3787553310394287, "sampling/importance_sampling_ratio/mean": 1.003217101097107, "sampling/importance_sampling_ratio/min": 0.673724889755249, "sampling/sampling_logp_difference/max": 0.39493346214294434, "sampling/sampling_logp_difference/mean": 0.011381209827959538, "step": 2369 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.013544891262426972, "clip_ratio/low_min": 0.013544891262426972, "clip_ratio/region_mean": 0.013544891262426972, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 34.5, "completions/mean_terminated_length": 34.5, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.21421233005821705, "epoch": 0.095192191830341, "frac_reward_zero_std": 0.0, "grad_norm": 12.456171035766602, "learning_rate": 2.821212121212121e-06, "loss": 0.0382, "num_tokens": 5364201.0, "reward": 0.9742438793182373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9742438793182373, "reward_meter_std": 0.03374186530709267, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03374185785651207, "reward_total_composite_mean": 0.9742438793182373, "reward_total_composite_std": 0.03374186530709267, "reward_total_mean": 0.9742438793182373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9742438793182373, "rewards/meter/std": 0.03374186530709267, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9742438793182373, "rewards/total_composite/std": 0.03374186530709267, "sampling/importance_sampling_ratio/max": 1.4533801078796387, "sampling/importance_sampling_ratio/mean": 1.0038917064666748, "sampling/importance_sampling_ratio/min": 0.34548693895339966, "sampling/sampling_logp_difference/max": 1.062800407409668, "sampling/sampling_logp_difference/mean": 0.02670416608452797, "step": 2370 }, { "clip_ratio/high_max": 0.0030120480805635452, "clip_ratio/high_mean": 0.0030120480805635452, "clip_ratio/low_mean": 0.0015060240402817726, "clip_ratio/low_min": 0.0015060240402817726, "clip_ratio/region_mean": 0.004518072120845318, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 83.0, "completions/mean_terminated_length": 83.0, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.030938314041122794, "epoch": 0.09523235731212595, "frac_reward_zero_std": 0.0, "grad_norm": 1.3944817781448364, "learning_rate": 2.818181818181818e-06, "loss": 0.0014, "num_tokens": 5366225.0, "reward": 0.9963405728340149, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9963405728340149, "reward_meter_std": 0.00012567358498927206, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00012567655357997864, "reward_total_composite_mean": 0.9963405728340149, "reward_total_composite_std": 0.00012567358498927206, "reward_total_mean": 0.9963405728340149, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9963405728340149, "rewards/meter/std": 0.00012567358498927206, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9963405728340149, "rewards/total_composite/std": 0.00012567358498927206, "sampling/importance_sampling_ratio/max": 1.3552405834197998, "sampling/importance_sampling_ratio/mean": 1.0000357627868652, "sampling/importance_sampling_ratio/min": 0.20796078443527222, "sampling/sampling_logp_difference/max": 1.5704057216644287, "sampling/sampling_logp_difference/mean": 0.006072253920137882, "step": 2371 }, { "clip_ratio/high_max": 0.03797508799470961, "clip_ratio/high_mean": 0.03797508799470961, "clip_ratio/low_mean": 0.0055555556900799274, "clip_ratio/low_min": 0.0055555556900799274, "clip_ratio/region_mean": 0.04353064368478954, "completions/clipped_ratio": 0.0, "completions/max_length": 135.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 132.5, "completions/mean_terminated_length": 132.5, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.38280507549643517, "epoch": 0.09527252279391091, "frac_reward_zero_std": 0.0, "grad_norm": 3.56820011138916, "learning_rate": 2.8151515151515153e-06, "loss": 0.0096, "num_tokens": 5368757.0, "reward": 0.9808491468429565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986804723739624, "reward_meter_std": 0.0002552097721491009, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.050487954169511795, "reward_total_composite_mean": 0.9808491468429565, "reward_total_composite_std": 0.0504879355430603, "reward_total_mean": 0.9808491468429565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986804723739624, "rewards/meter/std": 0.0002552097721491009, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9808491468429565, "rewards/total_composite/std": 0.0504879355430603, "sampling/importance_sampling_ratio/max": 1.8720965385437012, "sampling/importance_sampling_ratio/mean": 1.0080647468566895, "sampling/importance_sampling_ratio/min": 0.21261338889598846, "sampling/sampling_logp_difference/max": 1.5482797622680664, "sampling/sampling_logp_difference/mean": 0.04589386284351349, "step": 2372 }, { "clip_ratio/high_max": 0.02439746167510748, "clip_ratio/high_mean": 0.02439746167510748, "clip_ratio/low_mean": 0.00603448273614049, "clip_ratio/low_min": 0.00603448273614049, "clip_ratio/region_mean": 0.03043194441124797, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 143.625, "completions/mean_terminated_length": 143.625, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 0.2039720769971609, "epoch": 0.09531268827569586, "frac_reward_zero_std": 0.0, "grad_norm": 2.409925699234009, "learning_rate": 2.812121212121212e-06, "loss": 0.0088, "num_tokens": 5371394.0, "reward": 0.9634268879890442, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991071224212646, "reward_meter_std": 0.00023018992214929312, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06610530614852905, "reward_total_composite_mean": 0.9634268879890442, "reward_total_composite_std": 0.06610530614852905, "reward_total_mean": 0.9634268879890442, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991071224212646, "rewards/meter/std": 0.00023018992214929312, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9634268879890442, "rewards/total_composite/std": 0.06610530614852905, "sampling/importance_sampling_ratio/max": 1.691011667251587, "sampling/importance_sampling_ratio/mean": 1.0038185119628906, "sampling/importance_sampling_ratio/min": 0.24376478791236877, "sampling/sampling_logp_difference/max": 1.4115514755249023, "sampling/sampling_logp_difference/mean": 0.022872699424624443, "step": 2373 }, { "clip_ratio/high_max": 0.013676133356057107, "clip_ratio/high_mean": 0.013676133356057107, "clip_ratio/low_mean": 0.011739226174540818, "clip_ratio/low_min": 0.011739226174540818, "clip_ratio/region_mean": 0.025415359530597925, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 73.5, "completions/mean_terminated_length": 73.5, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.20485389418900013, "epoch": 0.09535285375748082, "frac_reward_zero_std": 0.0, "grad_norm": 1.9035519361495972, "learning_rate": 2.809090909090909e-06, "loss": 0.0065, "num_tokens": 5373190.0, "reward": 0.9992323517799377, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992323517799377, "reward_meter_std": 0.00020660254813265055, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020661571761593223, "reward_total_composite_mean": 0.9992323517799377, "reward_total_composite_std": 0.00020660254813265055, "reward_total_mean": 0.9992323517799377, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992323517799377, "rewards/meter/std": 0.00020660254813265055, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992323517799377, "rewards/total_composite/std": 0.00020660254813265055, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0032356977462769, "sampling/importance_sampling_ratio/min": 0.25946375727653503, "sampling/sampling_logp_difference/max": 1.3491382598876953, "sampling/sampling_logp_difference/mean": 0.031100856140255928, "step": 2374 }, { "clip_ratio/high_max": 0.006286898045800626, "clip_ratio/high_mean": 0.006286898045800626, "clip_ratio/low_mean": 0.00859655940439552, "clip_ratio/low_min": 0.00859655940439552, "clip_ratio/region_mean": 0.014883457450196147, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 319.625, "completions/mean_terminated_length": 319.625, "completions/min_length": 317.0, "completions/min_terminated_length": 317.0, "entropy": 0.20801045559346676, "epoch": 0.09539301923926577, "frac_reward_zero_std": 0.0, "grad_norm": 1.6295384168624878, "learning_rate": 2.806060606060606e-06, "loss": 0.0053, "num_tokens": 5377395.0, "reward": 0.7667578458786011, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988389015197754, "reward_meter_std": 0.0002653080737218261, "reward_repeat_penalty_mean": 0.8529411554336548, "reward_repeat_penalty_std": 0.03144249692559242, "reward_std": 0.02833147533237934, "reward_total_composite_mean": 0.7667578458786011, "reward_total_composite_std": 0.02833147719502449, "reward_total_mean": 0.7667578458786011, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988389015197754, "rewards/meter/std": 0.0002653080737218261, "rewards/repeat_penalty/mean": 0.8529411554336548, "rewards/repeat_penalty/std": 0.03144249692559242, "rewards/total_composite/mean": 0.7667578458786011, "rewards/total_composite/std": 0.02833147719502449, "sampling/importance_sampling_ratio/max": 1.622670292854309, "sampling/importance_sampling_ratio/mean": 1.0060769319534302, "sampling/importance_sampling_ratio/min": 0.26613834500312805, "sampling/sampling_logp_difference/max": 1.3237390518188477, "sampling/sampling_logp_difference/mean": 0.02020336128771305, "step": 2375 }, { "clip_ratio/high_max": 0.007490954245440662, "clip_ratio/high_mean": 0.007490954245440662, "clip_ratio/low_mean": 0.005597014795057476, "clip_ratio/low_min": 0.005597014795057476, "clip_ratio/region_mean": 0.013087969040498137, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.0682340869680047, "epoch": 0.09543318472105072, "frac_reward_zero_std": 0.0, "grad_norm": 0.18841330707073212, "learning_rate": 2.803030303030303e-06, "loss": 0.0009, "num_tokens": 5379314.0, "reward": 0.9981135129928589, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981135129928589, "reward_meter_std": 1.4713088603457436e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.4704202840221114e-05, "reward_total_composite_mean": 0.9981135129928589, "reward_total_composite_std": 1.4713088603457436e-05, "reward_total_mean": 0.9981135129928589, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981135129928589, "rewards/meter/std": 1.4713088603457436e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981135129928589, "rewards/total_composite/std": 1.4713088603457436e-05, "sampling/importance_sampling_ratio/max": 1.2641924619674683, "sampling/importance_sampling_ratio/mean": 0.9993737936019897, "sampling/importance_sampling_ratio/min": 0.40922650694847107, "sampling/sampling_logp_difference/max": 0.893486499786377, "sampling/sampling_logp_difference/mean": 0.010610595345497131, "step": 2376 }, { "clip_ratio/high_max": 0.005597014795057476, "clip_ratio/high_mean": 0.005597014795057476, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005597014795057476, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.06305565731599927, "epoch": 0.09547335020283568, "frac_reward_zero_std": 0.0, "grad_norm": 0.5763253569602966, "learning_rate": 2.8000000000000003e-06, "loss": -0.001, "num_tokens": 5381154.0, "reward": 0.9981157779693604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981157779693604, "reward_meter_std": 2.6794468794832937e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.6794450604938902e-05, "reward_total_composite_mean": 0.9981157779693604, "reward_total_composite_std": 2.6794468794832937e-05, "reward_total_mean": 0.9981157779693604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981157779693604, "rewards/meter/std": 2.6794468794832937e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981157779693604, "rewards/total_composite/std": 2.6794468794832937e-05, "sampling/importance_sampling_ratio/max": 1.3120979070663452, "sampling/importance_sampling_ratio/mean": 1.001448631286621, "sampling/importance_sampling_ratio/min": 0.6235338449478149, "sampling/sampling_logp_difference/max": 0.4723522663116455, "sampling/sampling_logp_difference/mean": 0.008252451196312904, "step": 2377 }, { "clip_ratio/high_max": 0.015454705455340445, "clip_ratio/high_mean": 0.015454705455340445, "clip_ratio/low_mean": 0.004003660404123366, "clip_ratio/low_min": 0.004003660404123366, "clip_ratio/region_mean": 0.01945836585946381, "completions/clipped_ratio": 0.0, "completions/max_length": 189.0, "completions/max_terminated_length": 189.0, "completions/mean_length": 186.125, "completions/mean_terminated_length": 186.125, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.2459548767656088, "epoch": 0.09551351568462063, "frac_reward_zero_std": 0.0, "grad_norm": 1.9172272682189941, "learning_rate": 2.7969696969696976e-06, "loss": 0.0034, "num_tokens": 5384219.0, "reward": 0.7978767156600952, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9751391410827637, "reward_meter_std": 0.04846389219164848, "reward_repeat_penalty_mean": 0.8181818723678589, "reward_repeat_penalty_std": 0.08416548371315002, "reward_std": 0.09277583658695221, "reward_total_composite_mean": 0.7978767156600952, "reward_total_composite_std": 0.09277582913637161, "reward_total_mean": 0.7978767156600952, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9751391410827637, "rewards/meter/std": 0.04846389219164848, "rewards/repeat_penalty/mean": 0.8181818723678589, "rewards/repeat_penalty/std": 0.08416548371315002, "rewards/total_composite/mean": 0.7978767156600952, "rewards/total_composite/std": 0.09277582913637161, "sampling/importance_sampling_ratio/max": 1.715010404586792, "sampling/importance_sampling_ratio/mean": 1.005282998085022, "sampling/importance_sampling_ratio/min": 0.3847850263118744, "sampling/sampling_logp_difference/max": 0.9550704956054688, "sampling/sampling_logp_difference/mean": 0.02777726948261261, "step": 2378 }, { "clip_ratio/high_max": 0.00916612590663135, "clip_ratio/high_mean": 0.00916612590663135, "clip_ratio/low_mean": 0.01117487601004541, "clip_ratio/low_min": 0.01117487601004541, "clip_ratio/region_mean": 0.02034100191667676, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.625, "completions/mean_terminated_length": 67.625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.32505345717072487, "epoch": 0.09555368116640559, "frac_reward_zero_std": 0.0, "grad_norm": 2.5564944744110107, "learning_rate": 2.7939393939393944e-06, "loss": 0.0033, "num_tokens": 5385952.0, "reward": 0.9991129636764526, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991129636764526, "reward_meter_std": 0.00021267922420520335, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00021267762349452823, "reward_total_composite_mean": 0.9991129636764526, "reward_total_composite_std": 0.00021267922420520335, "reward_total_mean": 0.9991129636764526, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991129636764526, "rewards/meter/std": 0.00021267922420520335, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991129636764526, "rewards/total_composite/std": 0.00021267922420520335, "sampling/importance_sampling_ratio/max": 1.5362749099731445, "sampling/importance_sampling_ratio/mean": 1.011904001235962, "sampling/importance_sampling_ratio/min": 0.5071040987968445, "sampling/sampling_logp_difference/max": 0.6790390014648438, "sampling/sampling_logp_difference/mean": 0.03177187591791153, "step": 2379 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.0037313431967049837, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.024645814672112465, "epoch": 0.09559384664819054, "frac_reward_zero_std": 0.0, "grad_norm": 0.42718541622161865, "learning_rate": 2.7909090909090912e-06, "loss": -0.0004, "num_tokens": 5387928.0, "reward": 0.9981181621551514, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981181621551514, "reward_meter_std": 2.501497874618508e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.5023065973073244e-05, "reward_total_composite_mean": 0.9981181621551514, "reward_total_composite_std": 2.501497874618508e-05, "reward_total_mean": 0.9981181621551514, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981181621551514, "rewards/meter/std": 2.501497874618508e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981181621551514, "rewards/total_composite/std": 2.501497874618508e-05, "sampling/importance_sampling_ratio/max": 1.1040713787078857, "sampling/importance_sampling_ratio/mean": 0.9986084699630737, "sampling/importance_sampling_ratio/min": 0.3541722297668457, "sampling/sampling_logp_difference/max": 1.0379719734191895, "sampling/sampling_logp_difference/mean": 0.006671510171145201, "step": 2380 }, { "clip_ratio/high_max": 0.02113948343321681, "clip_ratio/high_mean": 0.02113948343321681, "clip_ratio/low_mean": 0.0011261261533945799, "clip_ratio/low_min": 0.0011261261533945799, "clip_ratio/region_mean": 0.02226560958661139, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 107.25, "completions/mean_terminated_length": 107.25, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.1429230272769928, "epoch": 0.09563401212997549, "frac_reward_zero_std": 0.0, "grad_norm": 3.057466745376587, "learning_rate": 2.7878787878787885e-06, "loss": 0.0134, "num_tokens": 5390258.0, "reward": 0.9742231369018555, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992028474807739, "reward_meter_std": 7.249469490488991e-05, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07065972685813904, "reward_total_composite_mean": 0.9742231369018555, "reward_total_composite_std": 0.07065970450639725, "reward_total_mean": 0.9742231369018555, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992028474807739, "rewards/meter/std": 7.249469490488991e-05, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9742231369018555, "rewards/total_composite/std": 0.07065970450639725, "sampling/importance_sampling_ratio/max": 1.8616714477539062, "sampling/importance_sampling_ratio/mean": 1.0024255514144897, "sampling/importance_sampling_ratio/min": 0.4641132354736328, "sampling/sampling_logp_difference/max": 0.7676267623901367, "sampling/sampling_logp_difference/mean": 0.020719556137919426, "step": 2381 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.008781549287959933, "epoch": 0.09567417761176045, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.7848484848484853e-06, "loss": 0.0, "num_tokens": 5392402.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0541869401931763, "sampling/importance_sampling_ratio/mean": 1.0004812479019165, "sampling/importance_sampling_ratio/min": 0.9255428314208984, "sampling/sampling_logp_difference/max": 0.07737492024898529, "sampling/sampling_logp_difference/mean": 0.0010491692228242755, "step": 2382 }, { "clip_ratio/high_max": 0.014423077227547765, "clip_ratio/high_mean": 0.014423077227547765, "clip_ratio/low_mean": 0.0076776278438046575, "clip_ratio/low_min": 0.0076776278438046575, "clip_ratio/region_mean": 0.022100705071352422, "completions/clipped_ratio": 0.0, "completions/max_length": 131.0, "completions/max_terminated_length": 131.0, "completions/mean_length": 130.125, "completions/mean_terminated_length": 130.125, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.09555556159466505, "epoch": 0.0957143430935454, "frac_reward_zero_std": 0.0, "grad_norm": 1.7132611274719238, "learning_rate": 2.781818181818182e-06, "loss": 0.006, "num_tokens": 5394915.0, "reward": 0.9457492828369141, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999283492565155, "reward_meter_std": 2.7988780857413076e-05, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07386651635169983, "reward_total_composite_mean": 0.9457492828369141, "reward_total_composite_std": 0.07386652380228043, "reward_total_mean": 0.9457492828369141, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999283492565155, "rewards/meter/std": 2.7988780857413076e-05, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9457492828369141, "rewards/total_composite/std": 0.07386652380228043, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003693699836731, "sampling/importance_sampling_ratio/min": 0.2535138428211212, "sampling/sampling_logp_difference/max": 1.372336745262146, "sampling/sampling_logp_difference/mean": 0.016906656324863434, "step": 2383 }, { "clip_ratio/high_max": 0.03354567987844348, "clip_ratio/high_mean": 0.03354567987844348, "clip_ratio/low_mean": 0.01336255669593811, "clip_ratio/low_min": 0.01336255669593811, "clip_ratio/region_mean": 0.04690823657438159, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 101.125, "completions/mean_terminated_length": 101.125, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.41279827430844307, "epoch": 0.09575450857533035, "frac_reward_zero_std": 0.0, "grad_norm": 4.2915167808532715, "learning_rate": 2.778787878787879e-06, "loss": 0.0197, "num_tokens": 5397220.0, "reward": 0.9984184503555298, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984184503555298, "reward_meter_std": 0.0006285551935434341, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006285490817390382, "reward_total_composite_mean": 0.9984184503555298, "reward_total_composite_std": 0.0006285551935434341, "reward_total_mean": 0.9984184503555298, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984184503555298, "rewards/meter/std": 0.0006285551935434341, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984184503555298, "rewards/total_composite/std": 0.0006285551935434341, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017614364624023, "sampling/importance_sampling_ratio/min": 0.3605065941810608, "sampling/sampling_logp_difference/max": 1.02024507522583, "sampling/sampling_logp_difference/mean": 0.05373314768075943, "step": 2384 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.009345798986032605, "epoch": 0.09579467405711531, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.7757575757575762e-06, "loss": 0.0, "num_tokens": 5398716.0, "reward": 0.9980231523513794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9980231523513794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.024132490158081, "sampling/importance_sampling_ratio/mean": 1.0006725788116455, "sampling/importance_sampling_ratio/min": 0.9885799884796143, "sampling/sampling_logp_difference/max": 0.023845985531806946, "sampling/sampling_logp_difference/mean": 0.0007748155039735138, "step": 2385 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.004995487950509414, "epoch": 0.09583483953890026, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.772727272727273e-06, "loss": 0.0, "num_tokens": 5400332.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0083444118499756, "sampling/importance_sampling_ratio/mean": 1.0003135204315186, "sampling/importance_sampling_ratio/min": 0.9897130131721497, "sampling/sampling_logp_difference/max": 0.010340280830860138, "sampling/sampling_logp_difference/mean": 0.00041769127710722387, "step": 2386 }, { "clip_ratio/high_max": 0.019927536603063345, "clip_ratio/high_mean": 0.019927536603063345, "clip_ratio/low_mean": 0.016549831489101052, "clip_ratio/low_min": 0.016549831489101052, "clip_ratio/region_mean": 0.0364773680921644, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.371359147131443, "epoch": 0.09587500502068522, "frac_reward_zero_std": 0.0, "grad_norm": 3.3865015506744385, "learning_rate": 2.76969696969697e-06, "loss": 0.0021, "num_tokens": 5402149.0, "reward": 0.9987808465957642, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987808465957642, "reward_meter_std": 0.0002124947786796838, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00021250976715236902, "reward_total_composite_mean": 0.9987808465957642, "reward_total_composite_std": 0.0002124947786796838, "reward_total_mean": 0.9987808465957642, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987808465957642, "rewards/meter/std": 0.0002124947786796838, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987808465957642, "rewards/total_composite/std": 0.0002124947786796838, "sampling/importance_sampling_ratio/max": 1.5863909721374512, "sampling/importance_sampling_ratio/mean": 1.0035470724105835, "sampling/importance_sampling_ratio/min": 0.18070924282073975, "sampling/sampling_logp_difference/max": 1.7108659744262695, "sampling/sampling_logp_difference/mean": 0.05237334594130516, "step": 2387 }, { "clip_ratio/high_max": 0.00390625, "clip_ratio/high_mean": 0.00390625, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.00390625, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.015011629671789706, "epoch": 0.09591517050247017, "frac_reward_zero_std": 0.0, "grad_norm": 2.705258846282959, "learning_rate": 2.766666666666667e-06, "loss": 0.0005, "num_tokens": 5403597.0, "reward": 0.9992798566818237, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992798566818237, "reward_meter_std": 3.3992837416008115e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.3992837416008115e-05, "reward_total_composite_mean": 0.9992798566818237, "reward_total_composite_std": 3.3992837416008115e-05, "reward_total_mean": 0.9992798566818237, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992798566818237, "rewards/meter/std": 3.3992837416008115e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992798566818237, "rewards/total_composite/std": 3.3992837416008115e-05, "sampling/importance_sampling_ratio/max": 1.0776525735855103, "sampling/importance_sampling_ratio/mean": 0.9955343008041382, "sampling/importance_sampling_ratio/min": 0.3481365442276001, "sampling/sampling_logp_difference/max": 1.0551605224609375, "sampling/sampling_logp_difference/mean": 0.00878862477838993, "step": 2388 }, { "clip_ratio/high_max": 0.027370785945095122, "clip_ratio/high_mean": 0.027370785945095122, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/region_mean": 0.03451364312786609, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.3192524388432503, "epoch": 0.09595533598425512, "frac_reward_zero_std": 0.0, "grad_norm": 3.947572946548462, "learning_rate": 2.763636363636364e-06, "loss": 0.005, "num_tokens": 5405494.0, "reward": 0.9979345798492432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979345798492432, "reward_meter_std": 0.0023531855549663305, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0023531648330390453, "reward_total_composite_mean": 0.9979345798492432, "reward_total_composite_std": 0.0023531855549663305, "reward_total_mean": 0.9979345798492432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979345798492432, "rewards/meter/std": 0.0023531855549663305, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979345798492432, "rewards/total_composite/std": 0.0023531855549663305, "sampling/importance_sampling_ratio/max": 1.5146595239639282, "sampling/importance_sampling_ratio/mean": 1.0087910890579224, "sampling/importance_sampling_ratio/min": 0.3869292438030243, "sampling/sampling_logp_difference/max": 0.9495134353637695, "sampling/sampling_logp_difference/mean": 0.03587198257446289, "step": 2389 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 54.125, "completions/mean_terminated_length": 54.125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.004519450158113614, "epoch": 0.09599550146604008, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.760606060606061e-06, "loss": 0.0, "num_tokens": 5407279.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0198215246200562, "sampling/importance_sampling_ratio/mean": 0.9997290968894958, "sampling/importance_sampling_ratio/min": 0.841161847114563, "sampling/sampling_logp_difference/max": 0.17297124862670898, "sampling/sampling_logp_difference/mean": 0.0010611445177346468, "step": 2390 }, { "clip_ratio/high_max": 0.005099173518829048, "clip_ratio/high_mean": 0.005099173518829048, "clip_ratio/low_mean": 0.001033057807944715, "clip_ratio/low_min": 0.001033057807944715, "clip_ratio/region_mean": 0.006132231326773763, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 121.5, "completions/mean_terminated_length": 121.5, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.05755401449277997, "epoch": 0.09603566694782503, "frac_reward_zero_std": 0.0, "grad_norm": 1.7748810052871704, "learning_rate": 2.7575757575757576e-06, "loss": -0.0026, "num_tokens": 5409667.0, "reward": 0.9271951913833618, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9796151518821716, "reward_meter_std": 0.002762306947261095, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07331071048974991, "reward_total_composite_mean": 0.9271951913833618, "reward_total_composite_std": 0.0733107179403305, "reward_total_mean": 0.9271951913833618, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9796151518821716, "rewards/meter/std": 0.002762306947261095, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9271951913833618, "rewards/total_composite/std": 0.0733107179403305, "sampling/importance_sampling_ratio/max": 1.2646301984786987, "sampling/importance_sampling_ratio/mean": 1.0016850233078003, "sampling/importance_sampling_ratio/min": 0.2529063820838928, "sampling/sampling_logp_difference/max": 1.3747358322143555, "sampling/sampling_logp_difference/mean": 0.008518277667462826, "step": 2391 }, { "clip_ratio/high_max": 0.007223089050967246, "clip_ratio/high_mean": 0.007223089050967246, "clip_ratio/low_mean": 0.011445078765973449, "clip_ratio/low_min": 0.011445078765973449, "clip_ratio/region_mean": 0.018668167816940695, "completions/clipped_ratio": 0.0, "completions/max_length": 157.0, "completions/max_terminated_length": 157.0, "completions/mean_length": 154.125, "completions/mean_terminated_length": 154.125, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.23401772044599056, "epoch": 0.09607583242960999, "frac_reward_zero_std": 0.0, "grad_norm": 2.4178664684295654, "learning_rate": 2.754545454545455e-06, "loss": -0.002, "num_tokens": 5412340.0, "reward": 0.9002017974853516, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9878800511360168, "reward_meter_std": 0.02659641019999981, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.08359287679195404, "reward_total_composite_mean": 0.9002017974853516, "reward_total_composite_std": 0.08359287679195404, "reward_total_mean": 0.9002017974853516, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9878800511360168, "rewards/meter/std": 0.02659641019999981, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9002017974853516, "rewards/total_composite/std": 0.08359287679195404, "sampling/importance_sampling_ratio/max": 1.3937476873397827, "sampling/importance_sampling_ratio/mean": 1.0050873756408691, "sampling/importance_sampling_ratio/min": 0.20107191801071167, "sampling/sampling_logp_difference/max": 1.6040925979614258, "sampling/sampling_logp_difference/mean": 0.03138583526015282, "step": 2392 }, { "clip_ratio/high_max": 0.005208333372138441, "clip_ratio/high_mean": 0.005208333372138441, "clip_ratio/low_mean": 0.006779896095395088, "clip_ratio/low_min": 0.006779896095395088, "clip_ratio/region_mean": 0.011988229467533529, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 72.375, "completions/mean_terminated_length": 72.375, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.11769631877541542, "epoch": 0.09611599791139494, "frac_reward_zero_std": 0.0, "grad_norm": 1.324416995048523, "learning_rate": 2.7515151515151517e-06, "loss": 0.0047, "num_tokens": 5414319.0, "reward": 0.9992770552635193, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992770552635193, "reward_meter_std": 0.00018226148677058518, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00018225403618998826, "reward_total_composite_mean": 0.9992770552635193, "reward_total_composite_std": 0.00018226148677058518, "reward_total_mean": 0.9992770552635193, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992770552635193, "rewards/meter/std": 0.00018226148677058518, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992770552635193, "rewards/total_composite/std": 0.00018226148677058518, "sampling/importance_sampling_ratio/max": 1.7886258363723755, "sampling/importance_sampling_ratio/mean": 1.0025677680969238, "sampling/importance_sampling_ratio/min": 0.26453831791877747, "sampling/sampling_logp_difference/max": 1.3297691345214844, "sampling/sampling_logp_difference/mean": 0.020829232409596443, "step": 2393 }, { "clip_ratio/high_max": 0.0078125, "clip_ratio/high_mean": 0.0078125, "clip_ratio/low_mean": 0.01171875, "clip_ratio/low_min": 0.01171875, "clip_ratio/region_mean": 0.01953125, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.047464726492762566, "epoch": 0.09615616339317991, "frac_reward_zero_std": 0.0, "grad_norm": 1.3910056352615356, "learning_rate": 2.7484848484848486e-06, "loss": 0.0004, "num_tokens": 5416127.0, "reward": 0.9993087649345398, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993087649345398, "reward_meter_std": 5.955179949523881e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.955949382041581e-05, "reward_total_composite_mean": 0.9993087649345398, "reward_total_composite_std": 5.955179949523881e-05, "reward_total_mean": 0.9993087649345398, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993087649345398, "rewards/meter/std": 5.955179949523881e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993087649345398, "rewards/total_composite/std": 5.955179949523881e-05, "sampling/importance_sampling_ratio/max": 1.3544596433639526, "sampling/importance_sampling_ratio/mean": 0.9990496635437012, "sampling/importance_sampling_ratio/min": 0.4728599488735199, "sampling/sampling_logp_difference/max": 0.748956024646759, "sampling/sampling_logp_difference/mean": 0.012217044830322266, "step": 2394 }, { "clip_ratio/high_max": 0.0078125, "clip_ratio/high_mean": 0.0078125, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0078125, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.038509635254740715, "epoch": 0.09619632887496486, "frac_reward_zero_std": 0.0, "grad_norm": 0.06802152097225189, "learning_rate": 2.7454545454545454e-06, "loss": 0.0, "num_tokens": 5417895.0, "reward": 0.9993706345558167, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993706345558167, "reward_meter_std": 5.743691417592345e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.761006832472049e-06, "reward_total_composite_mean": 0.9993706345558167, "reward_total_composite_std": 5.743691417592345e-06, "reward_total_mean": 0.9993706345558167, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993706345558167, "rewards/meter/std": 5.743691417592345e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993706345558167, "rewards/total_composite/std": 5.743691417592345e-06, "sampling/importance_sampling_ratio/max": 1.6340101957321167, "sampling/importance_sampling_ratio/mean": 1.002130389213562, "sampling/importance_sampling_ratio/min": 0.5983215570449829, "sampling/sampling_logp_difference/max": 0.5136269330978394, "sampling/sampling_logp_difference/mean": 0.008546598255634308, "step": 2395 }, { "clip_ratio/high_max": 0.015861412626691163, "clip_ratio/high_mean": 0.015861412626691163, "clip_ratio/low_mean": 0.0012019231216982007, "clip_ratio/low_min": 0.0012019231216982007, "clip_ratio/region_mean": 0.017063335748389363, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 102.375, "completions/mean_terminated_length": 102.375, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.13659420423209667, "epoch": 0.09623649435674982, "frac_reward_zero_std": 0.0, "grad_norm": 1.6420077085494995, "learning_rate": 2.7424242424242426e-06, "loss": 0.0113, "num_tokens": 5420082.0, "reward": 0.968574583530426, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9934667944908142, "reward_meter_std": 0.006176711525768042, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06977621465921402, "reward_total_composite_mean": 0.968574583530426, "reward_total_composite_std": 0.06977622210979462, "reward_total_mean": 0.968574583530426, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9934667944908142, "rewards/meter/std": 0.006176711525768042, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.968574583530426, "rewards/total_composite/std": 0.06977622210979462, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0029125213623047, "sampling/importance_sampling_ratio/min": 0.38789549469947815, "sampling/sampling_logp_difference/max": 0.9480593204498291, "sampling/sampling_logp_difference/mean": 0.022047532722353935, "step": 2396 }, { "clip_ratio/high_max": 0.0028735632076859474, "clip_ratio/high_mean": 0.0028735632076859474, "clip_ratio/low_mean": 0.008893557591363788, "clip_ratio/low_min": 0.008893557591363788, "clip_ratio/region_mean": 0.011767120799049735, "completions/clipped_ratio": 0.0, "completions/max_length": 87.0, "completions/max_terminated_length": 87.0, "completions/mean_length": 84.5, "completions/mean_terminated_length": 84.5, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.06057103397324681, "epoch": 0.09627665983853477, "frac_reward_zero_std": 0.0, "grad_norm": 11.099533081054688, "learning_rate": 2.7393939393939395e-06, "loss": -0.0055, "num_tokens": 5422174.0, "reward": 0.9455694556236267, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9455694556236267, "reward_meter_std": 0.006192359142005444, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00619234936311841, "reward_total_composite_mean": 0.9455694556236267, "reward_total_composite_std": 0.006192359142005444, "reward_total_mean": 0.9455694556236267, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9455694556236267, "rewards/meter/std": 0.006192359142005444, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9455694556236267, "rewards/total_composite/std": 0.006192359142005444, "sampling/importance_sampling_ratio/max": 1.6855740547180176, "sampling/importance_sampling_ratio/mean": 0.9979245662689209, "sampling/importance_sampling_ratio/min": 0.36960431933403015, "sampling/sampling_logp_difference/max": 0.9953222274780273, "sampling/sampling_logp_difference/mean": 0.014308757148683071, "step": 2397 }, { "clip_ratio/high_max": 0.004206492216326296, "clip_ratio/high_mean": 0.004206492216326296, "clip_ratio/low_mean": 0.003401839407160878, "clip_ratio/low_min": 0.003401839407160878, "clip_ratio/region_mean": 0.0076083316234871745, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 147.875, "completions/mean_terminated_length": 147.875, "completions/min_length": 144.0, "completions/min_terminated_length": 144.0, "entropy": 0.04592153197154403, "epoch": 0.09631682532031972, "frac_reward_zero_std": 0.0, "grad_norm": 1.1203114986419678, "learning_rate": 2.7363636363636363e-06, "loss": -0.0017, "num_tokens": 5424845.0, "reward": 0.7985101938247681, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9743971824645996, "reward_meter_std": 0.0041354927234351635, "reward_repeat_penalty_mean": 0.819444477558136, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.05690954998135567, "reward_total_composite_mean": 0.7985101938247681, "reward_total_composite_std": 0.05690953880548477, "reward_total_mean": 0.7985101938247681, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9743971824645996, "rewards/meter/std": 0.0041354927234351635, "rewards/repeat_penalty/mean": 0.819444477558136, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.7985101938247681, "rewards/total_composite/std": 0.05690953880548477, "sampling/importance_sampling_ratio/max": 1.5559734106063843, "sampling/importance_sampling_ratio/mean": 0.9987753629684448, "sampling/importance_sampling_ratio/min": 0.20130351185798645, "sampling/sampling_logp_difference/max": 1.6029415130615234, "sampling/sampling_logp_difference/mean": 0.011587115004658699, "step": 2398 }, { "clip_ratio/high_max": 0.015463917283341289, "clip_ratio/high_mean": 0.015463917283341289, "clip_ratio/low_mean": 0.0013020833721384406, "clip_ratio/low_min": 0.0013020833721384406, "clip_ratio/region_mean": 0.01676600065547973, "completions/clipped_ratio": 0.0, "completions/max_length": 97.0, "completions/max_terminated_length": 97.0, "completions/mean_length": 96.875, "completions/mean_terminated_length": 96.875, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.05716710351407528, "epoch": 0.09635699080210468, "frac_reward_zero_std": 0.0, "grad_norm": 1.154660701751709, "learning_rate": 2.7333333333333336e-06, "loss": 0.0016, "num_tokens": 5427116.0, "reward": 0.9992877244949341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992877244949341, "reward_meter_std": 5.900250835111365e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.900421820115298e-05, "reward_total_composite_mean": 0.9992877244949341, "reward_total_composite_std": 5.900250835111365e-05, "reward_total_mean": 0.9992877244949341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992877244949341, "rewards/meter/std": 5.900250835111365e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992877244949341, "rewards/total_composite/std": 5.900250835111365e-05, "sampling/importance_sampling_ratio/max": 1.3556400537490845, "sampling/importance_sampling_ratio/mean": 0.9967014789581299, "sampling/importance_sampling_ratio/min": 0.27576398849487305, "sampling/sampling_logp_difference/max": 1.2882099151611328, "sampling/sampling_logp_difference/mean": 0.012875314801931381, "step": 2399 }, { "clip_ratio/high_max": 0.0034966744715347886, "clip_ratio/high_mean": 0.0034966744715347886, "clip_ratio/low_mean": 0.004999999888241291, "clip_ratio/low_min": 0.004999999888241291, "clip_ratio/region_mean": 0.00849667435977608, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 72.125, "completions/mean_terminated_length": 72.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.07446162821725011, "epoch": 0.09639715628388963, "frac_reward_zero_std": 0.0, "grad_norm": 1.3548955917358398, "learning_rate": 2.7303030303030304e-06, "loss": 0.0041, "num_tokens": 5428869.0, "reward": 0.9993604421615601, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993604421615601, "reward_meter_std": 9.307587606599554e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.308403969043866e-05, "reward_total_composite_mean": 0.9993604421615601, "reward_total_composite_std": 9.307587606599554e-05, "reward_total_mean": 0.9993604421615601, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993604421615601, "rewards/meter/std": 9.307587606599554e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993604421615601, "rewards/total_composite/std": 9.307587606599554e-05, "sampling/importance_sampling_ratio/max": 1.8390227556228638, "sampling/importance_sampling_ratio/mean": 1.0040256977081299, "sampling/importance_sampling_ratio/min": 0.32035067677497864, "sampling/sampling_logp_difference/max": 1.1383390426635742, "sampling/sampling_logp_difference/mean": 0.01761528290808201, "step": 2400 }, { "epoch": 0.09639715628388963, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 372.46153846153845, "eval_completions/max_terminated_length": 372.46153846153845, "eval_completions/mean_length": 196.8653846153846, "eval_completions/mean_terminated_length": 196.8653846153846, "eval_completions/min_length": 59.46153846153846, "eval_completions/min_terminated_length": 59.46153846153846, "eval_entropy": 0.19041176713429964, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 5428869.0, "eval_reward": 0.6122724092923678, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9244015262677119, "eval_reward_count_adherence_std": 0.09978143326365031, "eval_reward_meter_mean": 0.7441063798390902, "eval_reward_meter_std": 0.3844332993030548, "eval_reward_repeat_penalty_mean": 0.8776137416179364, "eval_reward_repeat_penalty_std": 0.1198913437815813, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6122724092923678, "eval_reward_total_composite_std": 0.35310536279128146, "eval_reward_total_mean": 0.6122724092923678, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9244015262677119, "eval_rewards/count_adherence/std": 0.09978143326365031, "eval_rewards/meter/mean": 0.7441063798390902, "eval_rewards/meter/std": 0.3844332993030548, "eval_rewards/repeat_penalty/mean": 0.8776137416179364, "eval_rewards/repeat_penalty/std": 0.1198913437815813, "eval_rewards/total_composite/mean": 0.6122724092923678, "eval_rewards/total_composite/std": 0.35310536279128146, "eval_runtime": 71.1062, "eval_samples_per_second": 1.463, "eval_sampling/importance_sampling_ratio/max": 1.4638345608344445, "eval_sampling/importance_sampling_ratio/mean": 1.004656663307777, "eval_sampling/importance_sampling_ratio/min": 0.36076818865079147, "eval_sampling/sampling_logp_difference/max": 1.0589995751014123, "eval_sampling/sampling_logp_difference/mean": 0.01914287731051445, "eval_steps_per_second": 0.183, "step": 2400 }, { "clip_ratio/high_max": 0.01538461574818939, "clip_ratio/high_mean": 0.01538461574818939, "clip_ratio/low_mean": 0.009360859869048, "clip_ratio/low_min": 0.009360859869048, "clip_ratio/region_mean": 0.02474547561723739, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 65.375, "completions/mean_terminated_length": 65.375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.1730100642889738, "epoch": 0.09643732176567459, "frac_reward_zero_std": 0.0, "grad_norm": 2.6642069816589355, "learning_rate": 2.7272727272727272e-06, "loss": 0.0072, "num_tokens": 5430680.0, "reward": 0.9385327100753784, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9385327100753784, "reward_meter_std": 0.10577936470508575, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.10577936470508575, "reward_total_composite_mean": 0.9385327100753784, "reward_total_composite_std": 0.10577936470508575, "reward_total_mean": 0.9385327100753784, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9385327100753784, "rewards/meter/std": 0.10577936470508575, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9385327100753784, "rewards/total_composite/std": 0.10577936470508575, "sampling/importance_sampling_ratio/max": 1.7399368286132812, "sampling/importance_sampling_ratio/mean": 1.0091955661773682, "sampling/importance_sampling_ratio/min": 0.56248939037323, "sampling/sampling_logp_difference/max": 0.5753829479217529, "sampling/sampling_logp_difference/mean": 0.02753767929971218, "step": 2401 }, { "clip_ratio/high_max": 0.028704919386655092, "clip_ratio/high_mean": 0.028704919386655092, "clip_ratio/low_mean": 0.007134926971048117, "clip_ratio/low_min": 0.007134926971048117, "clip_ratio/region_mean": 0.03583984635770321, "completions/clipped_ratio": 0.0, "completions/max_length": 90.0, "completions/max_terminated_length": 90.0, "completions/mean_length": 87.25, "completions/mean_terminated_length": 87.25, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.17476310580968857, "epoch": 0.09647748724745954, "frac_reward_zero_std": 0.0, "grad_norm": 5.85192346572876, "learning_rate": 2.724242424242424e-06, "loss": 0.0097, "num_tokens": 5432706.0, "reward": 0.9501332640647888, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9501332640647888, "reward_meter_std": 0.0800556018948555, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0800556018948555, "reward_total_composite_mean": 0.9501332640647888, "reward_total_composite_std": 0.0800556018948555, "reward_total_mean": 0.9501332640647888, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9501332640647888, "rewards/meter/std": 0.0800556018948555, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9501332640647888, "rewards/total_composite/std": 0.0800556018948555, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9965555667877197, "sampling/importance_sampling_ratio/min": 0.18733689188957214, "sampling/sampling_logp_difference/max": 1.6748466491699219, "sampling/sampling_logp_difference/mean": 0.03878026455640793, "step": 2402 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0028671720647253096, "epoch": 0.0965176527292445, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.7212121212121213e-06, "loss": 0.0, "num_tokens": 5434442.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0047893524169922, "sampling/importance_sampling_ratio/mean": 1.0003162622451782, "sampling/importance_sampling_ratio/min": 0.9962913393974304, "sampling/sampling_logp_difference/max": 0.004777892027050257, "sampling/sampling_logp_difference/mean": 0.00035840104101225734, "step": 2403 }, { "clip_ratio/high_max": 0.010291029699146748, "clip_ratio/high_mean": 0.010291029699146748, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.010291029699146748, "completions/clipped_ratio": 0.625, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 506.875, "completions/mean_terminated_length": 498.3333435058594, "completions/min_length": 489.0, "completions/min_terminated_length": 489.0, "entropy": 0.08459514752030373, "epoch": 0.09655781821102945, "frac_reward_zero_std": 0.0, "grad_norm": 0.281558632850647, "learning_rate": 2.718181818181818e-06, "loss": -0.1192, "num_tokens": 5437697.0, "reward": 0.6099640130996704, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6907894611358643, "reward_count_adherence_std": 0.01860806532204151, "reward_meter_mean": 0.9645444750785828, "reward_meter_std": 0.09066132456064224, "reward_repeat_penalty_mean": 0.9167022705078125, "reward_repeat_penalty_std": 0.039733175188302994, "reward_std": 0.057747457176446915, "reward_total_composite_mean": 0.6099640130996704, "reward_total_composite_std": 0.05774744227528572, "reward_total_mean": 0.6099640130996704, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6907894611358643, "rewards/count_adherence/std": 0.01860806532204151, "rewards/meter/mean": 0.9645444750785828, "rewards/meter/std": 0.09066132456064224, "rewards/repeat_penalty/mean": 0.9167022705078125, "rewards/repeat_penalty/std": 0.039733175188302994, "rewards/total_composite/mean": 0.6099640130996704, "rewards/total_composite/std": 0.05774744227528572, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0061771869659424, "sampling/importance_sampling_ratio/min": 0.07272414118051529, "sampling/sampling_logp_difference/max": 2.621081829071045, "sampling/sampling_logp_difference/mean": 0.03504529967904091, "step": 2404 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.003762025124160573, "epoch": 0.0965979836928144, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.715151515151516e-06, "loss": 0.0, "num_tokens": 5439465.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0083059072494507, "sampling/importance_sampling_ratio/mean": 1.0003942251205444, "sampling/importance_sampling_ratio/min": 0.9947773814201355, "sampling/sampling_logp_difference/max": 0.008271539583802223, "sampling/sampling_logp_difference/mean": 0.00042045500595122576, "step": 2405 }, { "clip_ratio/high_max": 0.005542142200283706, "clip_ratio/high_mean": 0.005542142200283706, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005542142200283706, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.375, "completions/mean_terminated_length": 67.375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.13846461195498705, "epoch": 0.09663814917459936, "frac_reward_zero_std": 0.0, "grad_norm": 4.301211357116699, "learning_rate": 2.7121212121212127e-06, "loss": 0.0059, "num_tokens": 5441252.0, "reward": 0.9776328802108765, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9776328802108765, "reward_meter_std": 0.05134515464305878, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.051345136016607285, "reward_total_composite_mean": 0.9776328802108765, "reward_total_composite_std": 0.05134515464305878, "reward_total_mean": 0.9776328802108765, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9776328802108765, "rewards/meter/std": 0.05134515464305878, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9776328802108765, "rewards/total_composite/std": 0.05134515464305878, "sampling/importance_sampling_ratio/max": 1.4865719079971313, "sampling/importance_sampling_ratio/mean": 1.0020570755004883, "sampling/importance_sampling_ratio/min": 0.3229783773422241, "sampling/sampling_logp_difference/max": 1.1301698684692383, "sampling/sampling_logp_difference/mean": 0.02041078545153141, "step": 2406 }, { "clip_ratio/high_max": 0.06410424876958132, "clip_ratio/high_mean": 0.06410424876958132, "clip_ratio/low_mean": 0.020202020648866892, "clip_ratio/low_min": 0.020202020648866892, "clip_ratio/region_mean": 0.08430626941844821, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.38669581711292267, "epoch": 0.09667831465638431, "frac_reward_zero_std": 0.0, "grad_norm": 13.151045799255371, "learning_rate": 2.7090909090909095e-06, "loss": 0.0436, "num_tokens": 5443171.0, "reward": 0.9381490349769592, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9381490349769592, "reward_meter_std": 0.009724577888846397, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009724575094878674, "reward_total_composite_mean": 0.9381490349769592, "reward_total_composite_std": 0.009724577888846397, "reward_total_mean": 0.9381490349769592, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9381490349769592, "rewards/meter/std": 0.009724577888846397, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9381490349769592, "rewards/total_composite/std": 0.009724577888846397, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9993854761123657, "sampling/importance_sampling_ratio/min": 0.12152271717786789, "sampling/sampling_logp_difference/max": 2.107654094696045, "sampling/sampling_logp_difference/mean": 0.08292688429355621, "step": 2407 }, { "clip_ratio/high_max": 0.01663423073478043, "clip_ratio/high_mean": 0.01663423073478043, "clip_ratio/low_mean": 0.002577319508418441, "clip_ratio/low_min": 0.002577319508418441, "clip_ratio/region_mean": 0.01921155024319887, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.75, "completions/mean_terminated_length": 97.75, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.04637799598276615, "epoch": 0.09671848013816926, "frac_reward_zero_std": 0.0, "grad_norm": 0.9304388165473938, "learning_rate": 2.7060606060606063e-06, "loss": -0.0002, "num_tokens": 5445313.0, "reward": 0.9989974498748779, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989974498748779, "reward_meter_std": 0.0009041029843501747, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009041179437190294, "reward_total_composite_mean": 0.9989974498748779, "reward_total_composite_std": 0.0009041029843501747, "reward_total_mean": 0.9989974498748779, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989974498748779, "rewards/meter/std": 0.0009041029843501747, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989974498748779, "rewards/total_composite/std": 0.0009041029843501747, "sampling/importance_sampling_ratio/max": 1.6353120803833008, "sampling/importance_sampling_ratio/mean": 0.9973183274269104, "sampling/importance_sampling_ratio/min": 0.2750520408153534, "sampling/sampling_logp_difference/max": 1.2907949686050415, "sampling/sampling_logp_difference/mean": 0.012694788165390491, "step": 2408 }, { "clip_ratio/high_max": 0.02023474802263081, "clip_ratio/high_mean": 0.02023474802263081, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/region_mean": 0.021858124644495547, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 154.125, "completions/mean_terminated_length": 154.125, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "entropy": 0.21369013004004955, "epoch": 0.09675864561995422, "frac_reward_zero_std": 0.0, "grad_norm": 2.528715133666992, "learning_rate": 2.7030303030303036e-06, "loss": 0.0069, "num_tokens": 5448370.0, "reward": 0.9619561433792114, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997596800327301, "reward_meter_std": 0.0010310488287359476, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06579122692346573, "reward_total_composite_mean": 0.9619561433792114, "reward_total_composite_std": 0.06579122692346573, "reward_total_mean": 0.9619561433792114, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997596800327301, "rewards/meter/std": 0.0010310488287359476, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9619561433792114, "rewards/total_composite/std": 0.06579122692346573, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004657506942749, "sampling/importance_sampling_ratio/min": 0.2534441351890564, "sampling/sampling_logp_difference/max": 1.3726117610931396, "sampling/sampling_logp_difference/mean": 0.028948156163096428, "step": 2409 }, { "clip_ratio/high_max": 0.016363060451112688, "clip_ratio/high_mean": 0.016363060451112688, "clip_ratio/low_mean": 0.007634902489371598, "clip_ratio/low_min": 0.007634902489371598, "clip_ratio/region_mean": 0.023997962940484285, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.1753720510751009, "epoch": 0.09679881110173917, "frac_reward_zero_std": 0.0, "grad_norm": 2.699066400527954, "learning_rate": 2.7000000000000004e-06, "loss": -0.0112, "num_tokens": 5450115.0, "reward": 0.9942950010299683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9942950010299683, "reward_meter_std": 0.0035084497649222612, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003508440451696515, "reward_total_composite_mean": 0.9942950010299683, "reward_total_composite_std": 0.0035084497649222612, "reward_total_mean": 0.9942950010299683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9942950010299683, "rewards/meter/std": 0.0035084497649222612, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9942950010299683, "rewards/total_composite/std": 0.0035084497649222612, "sampling/importance_sampling_ratio/max": 1.9520822763442993, "sampling/importance_sampling_ratio/mean": 1.0015225410461426, "sampling/importance_sampling_ratio/min": 0.35793569684028625, "sampling/sampling_logp_difference/max": 1.0274019241333008, "sampling/sampling_logp_difference/mean": 0.02465406060218811, "step": 2410 }, { "clip_ratio/high_max": 0.043241268722340465, "clip_ratio/high_mean": 0.043241268722340465, "clip_ratio/low_mean": 0.007401315961033106, "clip_ratio/low_min": 0.007401315961033106, "clip_ratio/region_mean": 0.05064258468337357, "completions/clipped_ratio": 0.0, "completions/max_length": 152.0, "completions/max_terminated_length": 152.0, "completions/mean_length": 135.375, "completions/mean_terminated_length": 135.375, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.38587315008044243, "epoch": 0.09683897658352413, "frac_reward_zero_std": 0.0, "grad_norm": 7.062148571014404, "learning_rate": 2.6969696969696972e-06, "loss": 0.0472, "num_tokens": 5452646.0, "reward": 0.9934861660003662, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9934861660003662, "reward_meter_std": 0.012767082080245018, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.012767070904374123, "reward_total_composite_mean": 0.9934861660003662, "reward_total_composite_std": 0.012767082080245018, "reward_total_mean": 0.9934861660003662, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9934861660003662, "rewards/meter/std": 0.012767082080245018, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934861660003662, "rewards/total_composite/std": 0.012767082080245018, "sampling/importance_sampling_ratio/max": 1.7682102918624878, "sampling/importance_sampling_ratio/mean": 1.0023579597473145, "sampling/importance_sampling_ratio/min": 0.2042321115732193, "sampling/sampling_logp_difference/max": 1.5884981155395508, "sampling/sampling_logp_difference/mean": 0.05575111508369446, "step": 2411 }, { "clip_ratio/high_max": 0.017077806871384382, "clip_ratio/high_mean": 0.017077806871384382, "clip_ratio/low_mean": 0.015032077208161354, "clip_ratio/low_min": 0.015032077208161354, "clip_ratio/region_mean": 0.032109884079545736, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 58.0, "completions/mean_terminated_length": 58.0, "completions/min_length": 55.0, "completions/min_terminated_length": 55.0, "entropy": 0.11028503353009, "epoch": 0.09687914206530908, "frac_reward_zero_std": 0.0, "grad_norm": 3.8710875511169434, "learning_rate": 2.6939393939393945e-06, "loss": 0.0204, "num_tokens": 5454382.0, "reward": 0.9953839778900146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953839778900146, "reward_meter_std": 0.0009903458412736654, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009903304744511843, "reward_total_composite_mean": 0.9953839778900146, "reward_total_composite_std": 0.0009903458412736654, "reward_total_mean": 0.9953839778900146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953839778900146, "rewards/meter/std": 0.0009903458412736654, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953839778900146, "rewards/total_composite/std": 0.0009903458412736654, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004859447479248, "sampling/importance_sampling_ratio/min": 0.44029319286346436, "sampling/sampling_logp_difference/max": 1.1877875328063965, "sampling/sampling_logp_difference/mean": 0.02391960285604, "step": 2412 }, { "clip_ratio/high_max": 0.02032158919610083, "clip_ratio/high_mean": 0.02032158919610083, "clip_ratio/low_mean": 0.006614301120862365, "clip_ratio/low_min": 0.006614301120862365, "clip_ratio/region_mean": 0.026935890316963196, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 345.375, "completions/mean_terminated_length": 345.375, "completions/min_length": 318.0, "completions/min_terminated_length": 318.0, "entropy": 0.2573400232940912, "epoch": 0.09691930754709403, "frac_reward_zero_std": 0.0, "grad_norm": 1.422156810760498, "learning_rate": 2.6909090909090913e-06, "loss": -0.0163, "num_tokens": 5459457.0, "reward": 0.8197740316390991, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.887499988079071, "reward_count_adherence_std": 0.035355325788259506, "reward_meter_mean": 0.9982728958129883, "reward_meter_std": 0.0010707140900194645, "reward_repeat_penalty_mean": 0.9245098233222961, "reward_repeat_penalty_std": 0.06273634731769562, "reward_std": 0.07304168492555618, "reward_total_composite_mean": 0.8197740316390991, "reward_total_composite_std": 0.07304169237613678, "reward_total_mean": 0.8197740316390991, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.887499988079071, "rewards/count_adherence/std": 0.035355325788259506, "rewards/meter/mean": 0.9982728958129883, "rewards/meter/std": 0.0010707140900194645, "rewards/repeat_penalty/mean": 0.9245098233222961, "rewards/repeat_penalty/std": 0.06273634731769562, "rewards/total_composite/mean": 0.8197740316390991, "rewards/total_composite/std": 0.07304169237613678, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0035998821258545, "sampling/importance_sampling_ratio/min": 0.3788796067237854, "sampling/sampling_logp_difference/max": 0.9705367088317871, "sampling/sampling_logp_difference/mean": 0.031854599714279175, "step": 2413 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.0018656715983524919, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.006271830061450601, "epoch": 0.09695947302887899, "frac_reward_zero_std": 0.0, "grad_norm": 0.009874632582068443, "learning_rate": 2.687878787878788e-06, "loss": -0.0, "num_tokens": 5461241.0, "reward": 0.9981490969657898, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981490969657898, "reward_meter_std": 8.935132427723147e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.935131518228445e-06, "reward_total_composite_mean": 0.9981490969657898, "reward_total_composite_std": 8.935132427723147e-06, "reward_total_mean": 0.9981490969657898, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981490969657898, "rewards/meter/std": 8.935132427723147e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981490969657898, "rewards/total_composite/std": 8.935132427723147e-06, "sampling/importance_sampling_ratio/max": 1.1797505617141724, "sampling/importance_sampling_ratio/mean": 1.0007779598236084, "sampling/importance_sampling_ratio/min": 0.980025053024292, "sampling/sampling_logp_difference/max": 0.16530299186706543, "sampling/sampling_logp_difference/mean": 0.0008906561997719109, "step": 2414 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.001466380010242574, "epoch": 0.09699963851066394, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.684848484848485e-06, "loss": 0.0, "num_tokens": 5462745.0, "reward": 0.9980231523513794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9980231523513794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0030980110168457, "sampling/importance_sampling_ratio/mean": 1.000183343887329, "sampling/importance_sampling_ratio/min": 0.9994993209838867, "sampling/sampling_logp_difference/max": 0.003093225881457329, "sampling/sampling_logp_difference/mean": 0.00018749816808849573, "step": 2415 }, { "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/region_mean": 0.0025510203558951616, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.023096754681319, "epoch": 0.0970398039924489, "frac_reward_zero_std": 0.0, "grad_norm": 0.24654512107372284, "learning_rate": 2.6818181818181822e-06, "loss": 0.0002, "num_tokens": 5464841.0, "reward": 0.9980032444000244, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980032444000244, "reward_meter_std": 8.327968316734768e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.322024768858682e-06, "reward_total_composite_mean": 0.9980032444000244, "reward_total_composite_std": 8.327968316734768e-06, "reward_total_mean": 0.9980032444000244, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980032444000244, "rewards/meter/std": 8.327968316734768e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980032444000244, "rewards/total_composite/std": 8.327968316734768e-06, "sampling/importance_sampling_ratio/max": 1.1473532915115356, "sampling/importance_sampling_ratio/mean": 1.0002275705337524, "sampling/importance_sampling_ratio/min": 0.45493748784065247, "sampling/sampling_logp_difference/max": 0.787595272064209, "sampling/sampling_logp_difference/mean": 0.0037318698596209288, "step": 2416 }, { "clip_ratio/high_max": 0.02460317499935627, "clip_ratio/high_mean": 0.02460317499935627, "clip_ratio/low_mean": 0.006756756920367479, "clip_ratio/low_min": 0.006756756920367479, "clip_ratio/region_mean": 0.03135993191972375, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 35.375, "completions/mean_terminated_length": 35.375, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.18990459106862545, "epoch": 0.09707996947423385, "frac_reward_zero_std": 0.0, "grad_norm": 3.3394386768341064, "learning_rate": 2.678787878787879e-06, "loss": 0.0088, "num_tokens": 5466420.0, "reward": 0.999369204044342, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999369204044342, "reward_meter_std": 0.0002697483287192881, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00026974434149451554, "reward_total_composite_mean": 0.999369204044342, "reward_total_composite_std": 0.0002697483287192881, "reward_total_mean": 0.999369204044342, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999369204044342, "rewards/meter/std": 0.0002697483287192881, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999369204044342, "rewards/total_composite/std": 0.0002697483287192881, "sampling/importance_sampling_ratio/max": 1.802677035331726, "sampling/importance_sampling_ratio/mean": 1.0062538385391235, "sampling/importance_sampling_ratio/min": 0.48778876662254333, "sampling/sampling_logp_difference/max": 0.7178728580474854, "sampling/sampling_logp_difference/mean": 0.02549952082335949, "step": 2417 }, { "clip_ratio/high_max": 0.05160611355677247, "clip_ratio/high_mean": 0.05160611355677247, "clip_ratio/low_mean": 0.012442129664123058, "clip_ratio/low_min": 0.012442129664123058, "clip_ratio/region_mean": 0.06404824322089553, "completions/clipped_ratio": 0.0, "completions/max_length": 85.0, "completions/max_terminated_length": 85.0, "completions/mean_length": 82.0, "completions/mean_terminated_length": 82.0, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.3536304421722889, "epoch": 0.0971201349560188, "frac_reward_zero_std": 0.0, "grad_norm": 11.102910041809082, "learning_rate": 2.675757575757576e-06, "loss": 0.0167, "num_tokens": 5468444.0, "reward": 0.8016860485076904, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8613059520721436, "reward_meter_std": 0.17161336541175842, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.1763620227575302, "reward_total_composite_mean": 0.8016860485076904, "reward_total_composite_std": 0.17636200785636902, "reward_total_mean": 0.8016860485076904, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8613059520721436, "rewards/meter/std": 0.17161336541175842, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8016860485076904, "rewards/total_composite/std": 0.17636200785636902, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9934130311012268, "sampling/importance_sampling_ratio/min": 0.07219072431325912, "sampling/sampling_logp_difference/max": 2.628443717956543, "sampling/sampling_logp_difference/mean": 0.08092934638261795, "step": 2418 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.00269945221953094, "epoch": 0.09716030043780376, "frac_reward_zero_std": 0.0, "grad_norm": 0.25810882449150085, "learning_rate": 2.6727272727272727e-06, "loss": -0.0001, "num_tokens": 5470156.0, "reward": 0.7876332998275757, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7876332998275757, "reward_meter_std": 1.5573279597447254e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.558229632792063e-05, "reward_total_composite_mean": 0.7876332998275757, "reward_total_composite_std": 1.5573279597447254e-05, "reward_total_mean": 0.7876332998275757, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7876332998275757, "rewards/meter/std": 1.5573279597447254e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876332998275757, "rewards/total_composite/std": 1.5573279597447254e-05, "sampling/importance_sampling_ratio/max": 1.006128191947937, "sampling/importance_sampling_ratio/mean": 0.9987244606018066, "sampling/importance_sampling_ratio/min": 0.3160611093044281, "sampling/sampling_logp_difference/max": 1.1518197059631348, "sampling/sampling_logp_difference/mean": 0.002996515017002821, "step": 2419 }, { "clip_ratio/high_max": 0.028763922629877925, "clip_ratio/high_mean": 0.028763922629877925, "clip_ratio/low_mean": 0.011130106868222356, "clip_ratio/low_min": 0.011130106868222356, "clip_ratio/region_mean": 0.03989402949810028, "completions/clipped_ratio": 0.0, "completions/max_length": 176.0, "completions/max_terminated_length": 176.0, "completions/mean_length": 169.125, "completions/mean_terminated_length": 169.125, "completions/min_length": 163.0, "completions/min_terminated_length": 163.0, "entropy": 0.3278270773589611, "epoch": 0.09720046591958871, "frac_reward_zero_std": 0.0, "grad_norm": 3.2308475971221924, "learning_rate": 2.66969696969697e-06, "loss": 0.005, "num_tokens": 5472925.0, "reward": 0.9572727680206299, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988895654678345, "reward_meter_std": 0.00021211641433183104, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.05750780552625656, "reward_total_composite_mean": 0.9572727680206299, "reward_total_composite_std": 0.057507775723934174, "reward_total_mean": 0.9572727680206299, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988895654678345, "rewards/meter/std": 0.00021211641433183104, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9572727680206299, "rewards/total_composite/std": 0.057507775723934174, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0061159133911133, "sampling/importance_sampling_ratio/min": 0.2250584363937378, "sampling/sampling_logp_difference/max": 1.4913952350616455, "sampling/sampling_logp_difference/mean": 0.04539603367447853, "step": 2420 }, { "clip_ratio/high_max": 0.004716981202363968, "clip_ratio/high_mean": 0.004716981202363968, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.004716981202363968, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 106.125, "completions/mean_terminated_length": 106.125, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.056452156975865364, "epoch": 0.09724063140137366, "frac_reward_zero_std": 0.0, "grad_norm": 0.6984437704086304, "learning_rate": 2.666666666666667e-06, "loss": -0.0028, "num_tokens": 5475014.0, "reward": 0.9992530345916748, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992530345916748, "reward_meter_std": 0.00013918116746935993, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001391761179547757, "reward_total_composite_mean": 0.9992530345916748, "reward_total_composite_std": 0.00013918116746935993, "reward_total_mean": 0.9992530345916748, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992530345916748, "rewards/meter/std": 0.00013918116746935993, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992530345916748, "rewards/total_composite/std": 0.00013918116746935993, "sampling/importance_sampling_ratio/max": 1.1507465839385986, "sampling/importance_sampling_ratio/mean": 1.000658392906189, "sampling/importance_sampling_ratio/min": 0.5418351888656616, "sampling/sampling_logp_difference/max": 0.6127934455871582, "sampling/sampling_logp_difference/mean": 0.008111930452287197, "step": 2421 }, { "clip_ratio/high_max": 0.012170514499302953, "clip_ratio/high_mean": 0.012170514499302953, "clip_ratio/low_mean": 0.010076848906464875, "clip_ratio/low_min": 0.010076848906464875, "clip_ratio/region_mean": 0.022247363405767828, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 134.75, "completions/mean_terminated_length": 134.75, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.23221887275576591, "epoch": 0.09728079688315862, "frac_reward_zero_std": 0.0, "grad_norm": 2.5043861865997314, "learning_rate": 2.6636363636363637e-06, "loss": 0.0142, "num_tokens": 5477412.0, "reward": 0.841211199760437, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9812095761299133, "reward_meter_std": 0.03474520519375801, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.1079898476600647, "reward_std": 0.1127474457025528, "reward_total_composite_mean": 0.841211199760437, "reward_total_composite_std": 0.1127474382519722, "reward_total_mean": 0.841211199760437, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9812095761299133, "rewards/meter/std": 0.03474520519375801, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.1079898476600647, "rewards/total_composite/mean": 0.841211199760437, "rewards/total_composite/std": 0.1127474382519722, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0124142169952393, "sampling/importance_sampling_ratio/min": 0.36081838607788086, "sampling/sampling_logp_difference/max": 1.1920127868652344, "sampling/sampling_logp_difference/mean": 0.026558201760053635, "step": 2422 }, { "clip_ratio/high_max": 0.009765625, "clip_ratio/high_mean": 0.009765625, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.009765625, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.125, "completions/mean_terminated_length": 64.125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.021533598424866796, "epoch": 0.09732096236494357, "frac_reward_zero_std": 0.0, "grad_norm": 5.193559169769287, "learning_rate": 2.660606060606061e-06, "loss": 0.0077, "num_tokens": 5479245.0, "reward": 0.9993478059768677, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993478059768677, "reward_meter_std": 0.0001315748959314078, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00013158157526049763, "reward_total_composite_mean": 0.9993478059768677, "reward_total_composite_std": 0.0001315748959314078, "reward_total_mean": 0.9993478059768677, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993478059768677, "rewards/meter/std": 0.0001315748959314078, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993478059768677, "rewards/total_composite/std": 0.0001315748959314078, "sampling/importance_sampling_ratio/max": 1.234817385673523, "sampling/importance_sampling_ratio/mean": 0.9980660080909729, "sampling/importance_sampling_ratio/min": 0.21038421988487244, "sampling/sampling_logp_difference/max": 1.5588197708129883, "sampling/sampling_logp_difference/mean": 0.006412571296095848, "step": 2423 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.022042680298909545, "clip_ratio/low_min": 0.022042680298909545, "clip_ratio/region_mean": 0.022042680298909545, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 187.5, "completions/mean_terminated_length": 79.33333587646484, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.1750364489853382, "epoch": 0.09736112784672853, "frac_reward_zero_std": 0.0, "grad_norm": 1.59493887424469, "learning_rate": 2.6575757575757577e-06, "loss": 0.1719, "num_tokens": 5481641.0, "reward": 0.21816782653331757, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.23749999701976776, "reward_count_adherence_std": 0.25460052490234375, "reward_meter_mean": 0.9973663687705994, "reward_meter_std": 0.0009385403245687485, "reward_repeat_penalty_mean": 0.9709615707397461, "reward_repeat_penalty_std": 0.06743928045034409, "reward_std": 0.22094321250915527, "reward_total_composite_mean": 0.21816782653331757, "reward_total_composite_std": 0.22094322741031647, "reward_total_mean": 0.21816782653331757, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.23749999701976776, "rewards/count_adherence/std": 0.25460052490234375, "rewards/meter/mean": 0.9973663687705994, "rewards/meter/std": 0.0009385403245687485, "rewards/repeat_penalty/mean": 0.9709615707397461, "rewards/repeat_penalty/std": 0.06743928045034409, "rewards/total_composite/mean": 0.21816782653331757, "rewards/total_composite/std": 0.22094322741031647, "sampling/importance_sampling_ratio/max": 1.734250783920288, "sampling/importance_sampling_ratio/mean": 0.99797523021698, "sampling/importance_sampling_ratio/min": 0.17430277168750763, "sampling/sampling_logp_difference/max": 1.7469613552093506, "sampling/sampling_logp_difference/mean": 0.041650205850601196, "step": 2424 }, { "clip_ratio/high_max": 0.017109742388129234, "clip_ratio/high_mean": 0.017109742388129234, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.02078621299006045, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 101.625, "completions/mean_terminated_length": 101.625, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.20133807510137558, "epoch": 0.09740129332851348, "frac_reward_zero_std": 0.0, "grad_norm": 3.781071186065674, "learning_rate": 2.6545454545454546e-06, "loss": 0.0094, "num_tokens": 5483726.0, "reward": 0.8708481192588806, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8708481192588806, "reward_meter_std": 0.34927064180374146, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.34927064180374146, "reward_total_composite_mean": 0.8708481192588806, "reward_total_composite_std": 0.34927064180374146, "reward_total_mean": 0.8708481192588806, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8708481192588806, "rewards/meter/std": 0.34927064180374146, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8708481192588806, "rewards/total_composite/std": 0.34927064180374146, "sampling/importance_sampling_ratio/max": 1.9378726482391357, "sampling/importance_sampling_ratio/mean": 1.0064235925674438, "sampling/importance_sampling_ratio/min": 0.40862902998924255, "sampling/sampling_logp_difference/max": 0.8949475288391113, "sampling/sampling_logp_difference/mean": 0.02785804308950901, "step": 2425 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.004820480738999322, "epoch": 0.09744145881029843, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.6515151515151514e-06, "loss": 0.0, "num_tokens": 5485422.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0289011001586914, "sampling/importance_sampling_ratio/mean": 1.0000972747802734, "sampling/importance_sampling_ratio/min": 0.9665557742118835, "sampling/sampling_logp_difference/max": 0.03401622548699379, "sampling/sampling_logp_difference/mean": 0.00044331722892820835, "step": 2426 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.011770866345614195, "epoch": 0.09748162429208339, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.6484848484848487e-06, "loss": 0.0, "num_tokens": 5487246.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0430419445037842, "sampling/importance_sampling_ratio/mean": 1.0001752376556396, "sampling/importance_sampling_ratio/min": 0.9197707772254944, "sampling/sampling_logp_difference/max": 0.0836307555437088, "sampling/sampling_logp_difference/mean": 0.0009743176633492112, "step": 2427 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.01067307055927813, "epoch": 0.09752178977386834, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.6454545454545455e-06, "loss": 0.0, "num_tokens": 5488686.0, "reward": 0.9943599104881287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9943599104881287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0674165487289429, "sampling/importance_sampling_ratio/mean": 0.9964240789413452, "sampling/importance_sampling_ratio/min": 0.4392264783382416, "sampling/sampling_logp_difference/max": 0.8227400779724121, "sampling/sampling_logp_difference/mean": 0.007659213151782751, "step": 2428 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.011805565096437931, "epoch": 0.0975619552556533, "frac_reward_zero_std": 0.0, "grad_norm": 0.09379827976226807, "learning_rate": 2.6424242424242423e-06, "loss": 0.0, "num_tokens": 5490494.0, "reward": 0.9993983507156372, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993983507156372, "reward_meter_std": 3.034573182958411e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.034573182958411e-06, "reward_total_composite_mean": 0.9993983507156372, "reward_total_composite_std": 3.034573182958411e-06, "reward_total_mean": 0.9993983507156372, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993983507156372, "rewards/meter/std": 3.034573182958411e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993983507156372, "rewards/total_composite/std": 3.034573182958411e-06, "sampling/importance_sampling_ratio/max": 1.1746667623519897, "sampling/importance_sampling_ratio/mean": 1.000643253326416, "sampling/importance_sampling_ratio/min": 0.9115949869155884, "sampling/sampling_logp_difference/max": 0.16098451614379883, "sampling/sampling_logp_difference/mean": 0.0012968340888619423, "step": 2429 }, { "clip_ratio/high_max": 0.02971602266188711, "clip_ratio/high_mean": 0.02971602266188711, "clip_ratio/low_mean": 0.009443168761208653, "clip_ratio/low_min": 0.009443168761208653, "clip_ratio/region_mean": 0.03915919142309576, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2827076967805624, "epoch": 0.09760212073743825, "frac_reward_zero_std": 0.0, "grad_norm": 5.21270751953125, "learning_rate": 2.6393939393939396e-06, "loss": 0.0007, "num_tokens": 5492246.0, "reward": 0.9988259077072144, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988259077072144, "reward_meter_std": 0.0006359845283441246, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006359879043884575, "reward_total_composite_mean": 0.9988259077072144, "reward_total_composite_std": 0.0006359845283441246, "reward_total_mean": 0.9988259077072144, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988259077072144, "rewards/meter/std": 0.0006359845283441246, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988259077072144, "rewards/total_composite/std": 0.0006359845283441246, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002232551574707, "sampling/importance_sampling_ratio/min": 0.16750246286392212, "sampling/sampling_logp_difference/max": 1.786757230758667, "sampling/sampling_logp_difference/mean": 0.05176051706075668, "step": 2430 }, { "clip_ratio/high_max": 0.023538942215964198, "clip_ratio/high_mean": 0.023538942215964198, "clip_ratio/low_mean": 0.019234189530834556, "clip_ratio/low_min": 0.019234189530834556, "clip_ratio/region_mean": 0.042773131746798754, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 102.25, "completions/mean_terminated_length": 102.25, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.3069701548665762, "epoch": 0.0976422862192232, "frac_reward_zero_std": 0.0, "grad_norm": 3.1840193271636963, "learning_rate": 2.6363636363636364e-06, "loss": 0.0117, "num_tokens": 5494352.0, "reward": 0.998919665813446, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998919665813446, "reward_meter_std": 0.00028825365006923676, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002882632543332875, "reward_total_composite_mean": 0.998919665813446, "reward_total_composite_std": 0.00028825365006923676, "reward_total_mean": 0.998919665813446, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998919665813446, "rewards/meter/std": 0.00028825365006923676, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998919665813446, "rewards/total_composite/std": 0.00028825365006923676, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0071525573730469, "sampling/importance_sampling_ratio/min": 0.3685198724269867, "sampling/sampling_logp_difference/max": 1.0248355865478516, "sampling/sampling_logp_difference/mean": 0.03998712822794914, "step": 2431 }, { "clip_ratio/high_max": 0.0035590012557804585, "clip_ratio/high_mean": 0.0035590012557804585, "clip_ratio/low_mean": 0.0015080645098350942, "clip_ratio/low_min": 0.0015080645098350942, "clip_ratio/region_mean": 0.005067065765615553, "completions/clipped_ratio": 0.0, "completions/max_length": 250.0, "completions/max_terminated_length": 250.0, "completions/mean_length": 246.875, "completions/mean_terminated_length": 246.875, "completions/min_length": 245.0, "completions/min_terminated_length": 245.0, "entropy": 0.08605087455362082, "epoch": 0.09768245170100816, "frac_reward_zero_std": 0.0, "grad_norm": 0.9189408421516418, "learning_rate": 2.6333333333333332e-06, "loss": 0.0053, "num_tokens": 5497991.0, "reward": 0.7301368713378906, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999138593673706, "reward_meter_std": 0.00011935058864764869, "reward_repeat_penalty_mean": 0.7307692766189575, "reward_repeat_penalty_std": 0.041117113083601, "reward_std": 0.041023530066013336, "reward_total_composite_mean": 0.7301368713378906, "reward_total_composite_std": 0.041023507714271545, "reward_total_mean": 0.7301368713378906, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999138593673706, "rewards/meter/std": 0.00011935058864764869, "rewards/repeat_penalty/mean": 0.7307692766189575, "rewards/repeat_penalty/std": 0.041117113083601, "rewards/total_composite/mean": 0.7301368713378906, "rewards/total_composite/std": 0.041023507714271545, "sampling/importance_sampling_ratio/max": 1.3559163808822632, "sampling/importance_sampling_ratio/mean": 1.0021779537200928, "sampling/importance_sampling_ratio/min": 0.3485677242279053, "sampling/sampling_logp_difference/max": 1.0539226531982422, "sampling/sampling_logp_difference/mean": 0.013350401073694229, "step": 2432 }, { "clip_ratio/high_max": 0.0014204545877873898, "clip_ratio/high_mean": 0.0014204545877873898, "clip_ratio/low_mean": 0.0021186440717428923, "clip_ratio/low_min": 0.0021186440717428923, "clip_ratio/region_mean": 0.003539098659530282, "completions/clipped_ratio": 0.0, "completions/max_length": 177.0, "completions/max_terminated_length": 177.0, "completions/mean_length": 176.875, "completions/mean_terminated_length": 176.875, "completions/min_length": 176.0, "completions/min_terminated_length": 176.0, "entropy": 0.05232963711023331, "epoch": 0.09772261718279311, "frac_reward_zero_std": 0.0, "grad_norm": 0.08106043934822083, "learning_rate": 2.63030303030303e-06, "loss": -0.0002, "num_tokens": 5500582.0, "reward": 0.7771967649459839, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992530345916748, "reward_meter_std": 1.491289003752172e-05, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_std": 1.1595264368224889e-05, "reward_total_composite_mean": 0.7771967649459839, "reward_total_composite_std": 1.1599345270951744e-05, "reward_total_mean": 0.7771967649459839, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992530345916748, "rewards/meter/std": 1.491289003752172e-05, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7771967649459839, "rewards/total_composite/std": 1.1599345270951744e-05, "sampling/importance_sampling_ratio/max": 1.2289493083953857, "sampling/importance_sampling_ratio/mean": 1.002611517906189, "sampling/importance_sampling_ratio/min": 0.4496315121650696, "sampling/sampling_logp_difference/max": 0.7993268966674805, "sampling/sampling_logp_difference/mean": 0.007231563795357943, "step": 2433 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.005529700662009418, "epoch": 0.09776278266457807, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.6272727272727278e-06, "loss": 0.0, "num_tokens": 5502654.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0090855360031128, "sampling/importance_sampling_ratio/mean": 1.0005393028259277, "sampling/importance_sampling_ratio/min": 0.9965153932571411, "sampling/sampling_logp_difference/max": 0.009044456295669079, "sampling/sampling_logp_difference/mean": 0.0005770151037722826, "step": 2434 }, { "clip_ratio/high_max": 0.03053560631815344, "clip_ratio/high_mean": 0.03053560631815344, "clip_ratio/low_mean": 0.00696005008649081, "clip_ratio/low_min": 0.00696005008649081, "clip_ratio/region_mean": 0.03749565640464425, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 90.125, "completions/mean_terminated_length": 90.125, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.2525391634553671, "epoch": 0.09780294814636302, "frac_reward_zero_std": 0.0, "grad_norm": 5.022347927093506, "learning_rate": 2.6242424242424246e-06, "loss": 0.003, "num_tokens": 5504823.0, "reward": 0.9594758749008179, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9843810200691223, "reward_meter_std": 0.03011130914092064, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07208065688610077, "reward_total_composite_mean": 0.9594758749008179, "reward_total_composite_std": 0.07208064943552017, "reward_total_mean": 0.9594758749008179, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9843810200691223, "rewards/meter/std": 0.03011130914092064, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9594758749008179, "rewards/total_composite/std": 0.07208064943552017, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028375387191772, "sampling/importance_sampling_ratio/min": 0.22001299262046814, "sampling/sampling_logp_difference/max": 1.5140687227249146, "sampling/sampling_logp_difference/mean": 0.03877629339694977, "step": 2435 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.004693651368143037, "epoch": 0.09784311362814797, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.621212121212122e-06, "loss": 0.0, "num_tokens": 5506655.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0094116926193237, "sampling/importance_sampling_ratio/mean": 1.0004562139511108, "sampling/importance_sampling_ratio/min": 0.9982850551605225, "sampling/sampling_logp_difference/max": 0.009367566555738449, "sampling/sampling_logp_difference/mean": 0.0004700902500189841, "step": 2436 }, { "clip_ratio/high_max": 0.003763133892789483, "clip_ratio/high_mean": 0.003763133892789483, "clip_ratio/low_mean": 0.00376287626568228, "clip_ratio/low_min": 0.00376287626568228, "clip_ratio/region_mean": 0.007526010158471763, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 100.25, "completions/mean_terminated_length": 100.25, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.04010937362909317, "epoch": 0.09788327910993293, "frac_reward_zero_std": 0.0, "grad_norm": 0.9422140717506409, "learning_rate": 2.6181818181818187e-06, "loss": 0.006, "num_tokens": 5508881.0, "reward": 0.9990655183792114, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990655183792114, "reward_meter_std": 0.00015276693738996983, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00015277891361620277, "reward_total_composite_mean": 0.9990655183792114, "reward_total_composite_std": 0.00015276693738996983, "reward_total_mean": 0.9990655183792114, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990655183792114, "rewards/meter/std": 0.00015276693738996983, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990655183792114, "rewards/total_composite/std": 0.00015276693738996983, "sampling/importance_sampling_ratio/max": 1.4387589693069458, "sampling/importance_sampling_ratio/mean": 0.9993890523910522, "sampling/importance_sampling_ratio/min": 0.36125504970550537, "sampling/sampling_logp_difference/max": 1.0181710720062256, "sampling/sampling_logp_difference/mean": 0.010102971456944942, "step": 2437 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/region_mean": 0.001923076924867928, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.25, "completions/mean_terminated_length": 64.25, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.010751945606898516, "epoch": 0.09792344459171788, "frac_reward_zero_std": 0.0, "grad_norm": 1.3743865489959717, "learning_rate": 2.6151515151515155e-06, "loss": 0.0011, "num_tokens": 5510611.0, "reward": 0.9993854761123657, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993854761123657, "reward_meter_std": 2.579814099590294e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.5807335987337865e-05, "reward_total_composite_mean": 0.9993854761123657, "reward_total_composite_std": 2.579814099590294e-05, "reward_total_mean": 0.9993854761123657, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993854761123657, "rewards/meter/std": 2.579814099590294e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993854761123657, "rewards/total_composite/std": 2.579814099590294e-05, "sampling/importance_sampling_ratio/max": 1.3340281248092651, "sampling/importance_sampling_ratio/mean": 1.0008904933929443, "sampling/importance_sampling_ratio/min": 0.5601092576980591, "sampling/sampling_logp_difference/max": 0.5796234607696533, "sampling/sampling_logp_difference/mean": 0.0030143873300403357, "step": 2438 }, { "clip_ratio/high_max": 0.02017652615904808, "clip_ratio/high_mean": 0.02017652615904808, "clip_ratio/low_mean": 0.01263557211495936, "clip_ratio/low_min": 0.01263557211495936, "clip_ratio/region_mean": 0.03281209827400744, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 98.625, "completions/mean_terminated_length": 98.625, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.282865421846509, "epoch": 0.09796361007350284, "frac_reward_zero_std": 0.0, "grad_norm": 3.004152774810791, "learning_rate": 2.6121212121212123e-06, "loss": 0.0054, "num_tokens": 5512728.0, "reward": 0.9990460872650146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990460872650146, "reward_meter_std": 0.00018491577066015452, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00018490130605641752, "reward_total_composite_mean": 0.9990460872650146, "reward_total_composite_std": 0.00018491577066015452, "reward_total_mean": 0.9990460872650146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990460872650146, "rewards/meter/std": 0.00018491577066015452, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990460872650146, "rewards/total_composite/std": 0.00018491577066015452, "sampling/importance_sampling_ratio/max": 1.664224624633789, "sampling/importance_sampling_ratio/mean": 1.0018714666366577, "sampling/importance_sampling_ratio/min": 0.2181643694639206, "sampling/sampling_logp_difference/max": 1.522506594657898, "sampling/sampling_logp_difference/mean": 0.039266087114810944, "step": 2439 }, { "clip_ratio/high_max": 0.01647980441339314, "clip_ratio/high_mean": 0.01647980441339314, "clip_ratio/low_mean": 0.00623403606005013, "clip_ratio/low_min": 0.00623403606005013, "clip_ratio/region_mean": 0.02271384047344327, "completions/clipped_ratio": 0.0, "completions/max_length": 358.0, "completions/max_terminated_length": 358.0, "completions/mean_length": 344.875, "completions/mean_terminated_length": 344.875, "completions/min_length": 329.0, "completions/min_terminated_length": 329.0, "entropy": 0.2484235055744648, "epoch": 0.09800377555528779, "frac_reward_zero_std": 0.0, "grad_norm": 1.5796713829040527, "learning_rate": 2.6090909090909096e-06, "loss": -0.0037, "num_tokens": 5517263.0, "reward": 0.7449136972427368, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8181818127632141, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985554218292236, "reward_meter_std": 0.00046954461140558124, "reward_repeat_penalty_mean": 0.9117647409439087, "reward_repeat_penalty_std": 0.04446640983223915, "reward_std": 0.036383435130119324, "reward_total_composite_mean": 0.7449136972427368, "reward_total_composite_std": 0.036383431404829025, "reward_total_mean": 0.7449136972427368, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8181818127632141, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985554218292236, "rewards/meter/std": 0.00046954461140558124, "rewards/repeat_penalty/mean": 0.9117647409439087, "rewards/repeat_penalty/std": 0.04446640983223915, "rewards/total_composite/mean": 0.7449136972427368, "rewards/total_composite/std": 0.036383431404829025, "sampling/importance_sampling_ratio/max": 1.8043537139892578, "sampling/importance_sampling_ratio/mean": 1.0063951015472412, "sampling/importance_sampling_ratio/min": 0.1864643543958664, "sampling/sampling_logp_difference/max": 1.6795151233673096, "sampling/sampling_logp_difference/mean": 0.030231278389692307, "step": 2440 }, { "clip_ratio/high_max": 0.025135348900221288, "clip_ratio/high_mean": 0.025135348900221288, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/region_mean": 0.028424822608940303, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 79.125, "completions/mean_terminated_length": 79.125, "completions/min_length": 76.0, "completions/min_terminated_length": 76.0, "entropy": 0.312020493671298, "epoch": 0.09804394103707274, "frac_reward_zero_std": 0.0, "grad_norm": 5.441035270690918, "learning_rate": 2.6060606060606064e-06, "loss": -0.0108, "num_tokens": 5519192.0, "reward": 0.8725272417068481, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9938977360725403, "reward_meter_std": 0.009339890442788601, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35255616903305054, "reward_total_composite_mean": 0.8725272417068481, "reward_total_composite_std": 0.35255616903305054, "reward_total_mean": 0.8725272417068481, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9938977360725403, "rewards/meter/std": 0.009339890442788601, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8725272417068481, "rewards/total_composite/std": 0.35255616903305054, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9990738034248352, "sampling/importance_sampling_ratio/min": 0.1210627406835556, "sampling/sampling_logp_difference/max": 2.1114463806152344, "sampling/sampling_logp_difference/mean": 0.03632466122508049, "step": 2441 }, { "clip_ratio/high_max": 0.0034722222480922937, "clip_ratio/high_mean": 0.0034722222480922937, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0034722222480922937, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 71.625, "completions/mean_terminated_length": 71.625, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.06646110257133842, "epoch": 0.0980841065188577, "frac_reward_zero_std": 0.0, "grad_norm": 1.337915062904358, "learning_rate": 2.6030303030303033e-06, "loss": -0.0071, "num_tokens": 5521213.0, "reward": 0.9993535876274109, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993535876274109, "reward_meter_std": 0.00013377754657994956, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00013379388838075101, "reward_total_composite_mean": 0.9993535876274109, "reward_total_composite_std": 0.00013377754657994956, "reward_total_mean": 0.9993535876274109, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993535876274109, "rewards/meter/std": 0.00013377754657994956, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993535876274109, "rewards/total_composite/std": 0.00013377754657994956, "sampling/importance_sampling_ratio/max": 1.1496003866195679, "sampling/importance_sampling_ratio/mean": 0.9988890290260315, "sampling/importance_sampling_ratio/min": 0.3751547038555145, "sampling/sampling_logp_difference/max": 0.9804167747497559, "sampling/sampling_logp_difference/mean": 0.013958643190562725, "step": 2442 }, { "clip_ratio/high_max": 0.026477832812815905, "clip_ratio/high_mean": 0.026477832812815905, "clip_ratio/low_mean": 0.04029961163178086, "clip_ratio/low_min": 0.04029961163178086, "clip_ratio/region_mean": 0.06677744444459677, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 97.375, "completions/mean_terminated_length": 97.375, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.3307408094406128, "epoch": 0.09812427200064265, "frac_reward_zero_std": 0.0, "grad_norm": 7.693568229675293, "learning_rate": 2.6e-06, "loss": 0.1046, "num_tokens": 5523328.0, "reward": 0.729360044002533, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.84375, "reward_count_adherence_std": 0.12938730418682098, "reward_meter_mean": 0.9511677026748657, "reward_meter_std": 0.009777599945664406, "reward_repeat_penalty_mean": 0.908730149269104, "reward_repeat_penalty_std": 0.08302231132984161, "reward_std": 0.13022364675998688, "reward_total_composite_mean": 0.729360044002533, "reward_total_composite_std": 0.13022364675998688, "reward_total_mean": 0.729360044002533, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.84375, "rewards/count_adherence/std": 0.12938730418682098, "rewards/meter/mean": 0.9511677026748657, "rewards/meter/std": 0.009777599945664406, "rewards/repeat_penalty/mean": 0.908730149269104, "rewards/repeat_penalty/std": 0.08302231132984161, "rewards/total_composite/mean": 0.729360044002533, "rewards/total_composite/std": 0.13022364675998688, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003010630607605, "sampling/importance_sampling_ratio/min": 0.06037522852420807, "sampling/sampling_logp_difference/max": 2.807176351547241, "sampling/sampling_logp_difference/mean": 0.06830313801765442, "step": 2443 }, { "clip_ratio/high_max": 0.006253673695027828, "clip_ratio/high_mean": 0.006253673695027828, "clip_ratio/low_mean": 0.0039014052599668503, "clip_ratio/low_min": 0.0039014052599668503, "clip_ratio/region_mean": 0.010155078954994678, "completions/clipped_ratio": 0.0, "completions/max_length": 321.0, "completions/max_terminated_length": 321.0, "completions/mean_length": 320.125, "completions/mean_terminated_length": 320.125, "completions/min_length": 319.0, "completions/min_terminated_length": 319.0, "entropy": 0.12365446705371141, "epoch": 0.0981644374824276, "frac_reward_zero_std": 0.0, "grad_norm": 1.5168893337249756, "learning_rate": 2.5969696969696973e-06, "loss": 0.0042, "num_tokens": 5527625.0, "reward": 0.74193274974823, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990435242652893, "reward_meter_std": 0.00013361620949581265, "reward_repeat_penalty_mean": 0.7426470518112183, "reward_repeat_penalty_std": 0.062391772866249084, "reward_std": 0.0622798316180706, "reward_total_composite_mean": 0.74193274974823, "reward_total_composite_std": 0.0622798353433609, "reward_total_mean": 0.74193274974823, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990435242652893, "rewards/meter/std": 0.00013361620949581265, "rewards/repeat_penalty/mean": 0.7426470518112183, "rewards/repeat_penalty/std": 0.062391772866249084, "rewards/total_composite/mean": 0.74193274974823, "rewards/total_composite/std": 0.0622798353433609, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.005713939666748, "sampling/importance_sampling_ratio/min": 0.41537001729011536, "sampling/sampling_logp_difference/max": 1.009657382965088, "sampling/sampling_logp_difference/mean": 0.016514526680111885, "step": 2444 }, { "clip_ratio/high_max": 0.02955011453013867, "clip_ratio/high_mean": 0.02955011453013867, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/region_mean": 0.03649455902632326, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2945162560790777, "epoch": 0.09820460296421256, "frac_reward_zero_std": 0.0, "grad_norm": 8.448700904846191, "learning_rate": 2.593939393939394e-06, "loss": 0.0227, "num_tokens": 5529594.0, "reward": 0.9891185760498047, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9891185760498047, "reward_meter_std": 0.028624890372157097, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.028624894097447395, "reward_total_composite_mean": 0.9891185760498047, "reward_total_composite_std": 0.028624890372157097, "reward_total_mean": 0.9891185760498047, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9891185760498047, "rewards/meter/std": 0.028624890372157097, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9891185760498047, "rewards/total_composite/std": 0.028624890372157097, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9979510307312012, "sampling/importance_sampling_ratio/min": 0.3536478877067566, "sampling/sampling_logp_difference/max": 1.0394535064697266, "sampling/sampling_logp_difference/mean": 0.04746231809258461, "step": 2445 }, { "clip_ratio/high_max": 0.003515693824738264, "clip_ratio/high_mean": 0.003515693824738264, "clip_ratio/low_mean": 0.0034616037737578154, "clip_ratio/low_min": 0.0034616037737578154, "clip_ratio/region_mean": 0.0069772975984960794, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 107.25, "completions/mean_terminated_length": 107.25, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.07135625509545207, "epoch": 0.09824476844599751, "frac_reward_zero_std": 0.0, "grad_norm": 0.9163700938224792, "learning_rate": 2.590909090909091e-06, "loss": 0.0023, "num_tokens": 5531756.0, "reward": 0.9992789626121521, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992789626121521, "reward_meter_std": 9.53831258811988e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.538961603539065e-05, "reward_total_composite_mean": 0.9992789626121521, "reward_total_composite_std": 9.53831258811988e-05, "reward_total_mean": 0.9992789626121521, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992789626121521, "rewards/meter/std": 9.53831258811988e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992789626121521, "rewards/total_composite/std": 9.53831258811988e-05, "sampling/importance_sampling_ratio/max": 1.2234734296798706, "sampling/importance_sampling_ratio/mean": 1.0028537511825562, "sampling/importance_sampling_ratio/min": 0.4823862612247467, "sampling/sampling_logp_difference/max": 0.7290101051330566, "sampling/sampling_logp_difference/mean": 0.010050991550087929, "step": 2446 }, { "clip_ratio/high_max": 0.024517936864867806, "clip_ratio/high_mean": 0.024517936864867806, "clip_ratio/low_mean": 0.009800048777833581, "clip_ratio/low_min": 0.009800048777833581, "clip_ratio/region_mean": 0.03431798564270139, "completions/clipped_ratio": 0.0, "completions/max_length": 472.0, "completions/max_terminated_length": 472.0, "completions/mean_length": 447.5, "completions/mean_terminated_length": 447.5, "completions/min_length": 421.0, "completions/min_terminated_length": 421.0, "entropy": 0.3355287965387106, "epoch": 0.09828493392778247, "frac_reward_zero_std": 0.0, "grad_norm": 1.6099560260772705, "learning_rate": 2.5878787878787883e-06, "loss": -0.0107, "num_tokens": 5537808.0, "reward": 0.7178632020950317, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7583333253860474, "reward_count_adherence_std": 0.0345032773911953, "reward_meter_mean": 0.9982467889785767, "reward_meter_std": 0.0005631654057651758, "reward_repeat_penalty_mean": 0.9484989643096924, "reward_repeat_penalty_std": 0.03878781571984291, "reward_std": 0.040422506630420685, "reward_total_composite_mean": 0.7178632020950317, "reward_total_composite_std": 0.040422506630420685, "reward_total_mean": 0.7178632020950317, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7583333253860474, "rewards/count_adherence/std": 0.0345032773911953, "rewards/meter/mean": 0.9982467889785767, "rewards/meter/std": 0.0005631654057651758, "rewards/repeat_penalty/mean": 0.9484989643096924, "rewards/repeat_penalty/std": 0.03878781571984291, "rewards/total_composite/mean": 0.7178632020950317, "rewards/total_composite/std": 0.040422506630420685, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003982424736023, "sampling/importance_sampling_ratio/min": 0.17287284135818481, "sampling/sampling_logp_difference/max": 1.7551989555358887, "sampling/sampling_logp_difference/mean": 0.041198987513780594, "step": 2447 }, { "clip_ratio/high_max": 0.01969089114572853, "clip_ratio/high_mean": 0.01969089114572853, "clip_ratio/low_mean": 0.00553075410425663, "clip_ratio/low_min": 0.00553075410425663, "clip_ratio/region_mean": 0.02522164524998516, "completions/clipped_ratio": 0.0, "completions/max_length": 320.0, "completions/max_terminated_length": 320.0, "completions/mean_length": 314.75, "completions/mean_terminated_length": 314.75, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "entropy": 0.26868183352053165, "epoch": 0.09832509940956742, "frac_reward_zero_std": 0.0, "grad_norm": 1.5874656438827515, "learning_rate": 2.584848484848485e-06, "loss": 0.0097, "num_tokens": 5541790.0, "reward": 0.9320781230926514, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986426830291748, "reward_meter_std": 0.0003120753972325474, "reward_repeat_penalty_mean": 0.9333333373069763, "reward_repeat_penalty_std": 0.07126966118812561, "reward_std": 0.0713275596499443, "reward_total_composite_mean": 0.9320781230926514, "reward_total_composite_std": 0.0713275596499443, "reward_total_mean": 0.9320781230926514, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986426830291748, "rewards/meter/std": 0.0003120753972325474, "rewards/repeat_penalty/mean": 0.9333333373069763, "rewards/repeat_penalty/std": 0.07126966118812561, "rewards/total_composite/mean": 0.9320781230926514, "rewards/total_composite/std": 0.0713275596499443, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0063222646713257, "sampling/importance_sampling_ratio/min": 0.34382739663124084, "sampling/sampling_logp_difference/max": 1.0676155090332031, "sampling/sampling_logp_difference/mean": 0.031515371054410934, "step": 2448 }, { "clip_ratio/high_max": 0.002396166091784835, "clip_ratio/high_mean": 0.002396166091784835, "clip_ratio/low_mean": 0.0027777778159361333, "clip_ratio/low_min": 0.0027777778159361333, "clip_ratio/region_mean": 0.005173943907720968, "completions/clipped_ratio": 0.0, "completions/max_length": 315.0, "completions/max_terminated_length": 315.0, "completions/mean_length": 314.75, "completions/mean_terminated_length": 314.75, "completions/min_length": 313.0, "completions/min_terminated_length": 313.0, "entropy": 0.07688481360673904, "epoch": 0.09836526489135237, "frac_reward_zero_std": 0.0, "grad_norm": 2.10903000831604, "learning_rate": 2.581818181818182e-06, "loss": 0.0056, "num_tokens": 5546236.0, "reward": 0.6889257431030273, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972984790802002, "reward_meter_std": 7.734992686891928e-05, "reward_repeat_penalty_mean": 0.6907894611358643, "reward_repeat_penalty_std": 0.06560124456882477, "reward_std": 0.06546018272638321, "reward_total_composite_mean": 0.6889257431030273, "reward_total_composite_std": 0.0654601976275444, "reward_total_mean": 0.6889257431030273, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972984790802002, "rewards/meter/std": 7.734992686891928e-05, "rewards/repeat_penalty/mean": 0.6907894611358643, "rewards/repeat_penalty/std": 0.06560124456882477, "rewards/total_composite/mean": 0.6889257431030273, "rewards/total_composite/std": 0.0654601976275444, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0018553733825684, "sampling/importance_sampling_ratio/min": 0.141923725605011, "sampling/sampling_logp_difference/max": 1.952465534210205, "sampling/sampling_logp_difference/mean": 0.014586514793336391, "step": 2449 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.004708475491497666, "epoch": 0.09840543037313733, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.5787878787878788e-06, "loss": 0.0, "num_tokens": 5547852.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0135822296142578, "sampling/importance_sampling_ratio/mean": 1.000256061553955, "sampling/importance_sampling_ratio/min": 0.9342393279075623, "sampling/sampling_logp_difference/max": 0.06802265346050262, "sampling/sampling_logp_difference/mean": 0.0006953292759135365, "step": 2450 }, { "epoch": 0.09840543037313733, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 394.38461538461536, "eval_completions/max_terminated_length": 394.38461538461536, "eval_completions/mean_length": 203.57692307692307, "eval_completions/mean_terminated_length": 203.57692307692307, "eval_completions/min_length": 61.38461538461539, "eval_completions/min_terminated_length": 61.38461538461539, "eval_entropy": 0.23695378464001876, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 5547852.0, "eval_reward": 0.6300999980706435, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9357927395747259, "eval_reward_count_adherence_std": 0.08574442823345844, "eval_reward_meter_mean": 0.7445399669500498, "eval_reward_meter_std": 0.3695275445397084, "eval_reward_repeat_penalty_mean": 0.8948483146153964, "eval_reward_repeat_penalty_std": 0.11098527220579293, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6300999980706435, "eval_reward_total_composite_std": 0.3479128239246515, "eval_reward_total_mean": 0.6300999980706435, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9357927395747259, "eval_rewards/count_adherence/std": 0.08574442823345844, "eval_rewards/meter/mean": 0.7445399669500498, "eval_rewards/meter/std": 0.3695275445397084, "eval_rewards/repeat_penalty/mean": 0.8948483146153964, "eval_rewards/repeat_penalty/std": 0.11098527220579293, "eval_rewards/total_composite/mean": 0.6300999980706435, "eval_rewards/total_composite/std": 0.3479128239246515, "eval_runtime": 74.6804, "eval_samples_per_second": 1.393, "eval_sampling/importance_sampling_ratio/max": 1.4525978748614972, "eval_sampling/importance_sampling_ratio/mean": 1.0057257138765776, "eval_sampling/importance_sampling_ratio/min": 0.37165725231170654, "eval_sampling/sampling_logp_difference/max": 0.9992542266845703, "eval_sampling/sampling_logp_difference/mean": 0.022716052543658476, "eval_steps_per_second": 0.174, "step": 2450 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.005932128522545099, "epoch": 0.09844559585492228, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.575757575757576e-06, "loss": 0.0, "num_tokens": 5549564.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0180944204330444, "sampling/importance_sampling_ratio/mean": 1.000001311302185, "sampling/importance_sampling_ratio/min": 0.8267950415611267, "sampling/sampling_logp_difference/max": 0.19019845128059387, "sampling/sampling_logp_difference/mean": 0.0012545905774459243, "step": 2451 }, { "clip_ratio/high_max": 0.01974922022782266, "clip_ratio/high_mean": 0.01974922022782266, "clip_ratio/low_mean": 0.013440471375361085, "clip_ratio/low_min": 0.013440471375361085, "clip_ratio/region_mean": 0.033189691603183746, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 92.875, "completions/mean_terminated_length": 92.875, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.2788791488856077, "epoch": 0.09848576133670724, "frac_reward_zero_std": 0.0, "grad_norm": 8.379902839660645, "learning_rate": 2.572727272727273e-06, "loss": 0.1072, "num_tokens": 5551659.0, "reward": 0.8898848295211792, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_meter_mean": 0.9926849007606506, "reward_meter_std": 0.0018546542851254344, "reward_repeat_penalty_mean": 0.9321428537368774, "reward_repeat_penalty_std": 0.09529759734869003, "reward_std": 0.15836970508098602, "reward_total_composite_mean": 0.8898848295211792, "reward_total_composite_std": 0.1583697348833084, "reward_total_mean": 0.8898848295211792, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/meter/mean": 0.9926849007606506, "rewards/meter/std": 0.0018546542851254344, "rewards/repeat_penalty/mean": 0.9321428537368774, "rewards/repeat_penalty/std": 0.09529759734869003, "rewards/total_composite/mean": 0.8898848295211792, "rewards/total_composite/std": 0.1583697348833084, "sampling/importance_sampling_ratio/max": 1.9530335664749146, "sampling/importance_sampling_ratio/mean": 1.0090692043304443, "sampling/importance_sampling_ratio/min": 0.27936431765556335, "sampling/sampling_logp_difference/max": 1.2752385139465332, "sampling/sampling_logp_difference/mean": 0.038993846625089645, "step": 2452 }, { "clip_ratio/high_max": 0.036513578495942056, "clip_ratio/high_mean": 0.036513578495942056, "clip_ratio/low_mean": 0.004360465332865715, "clip_ratio/low_min": 0.004360465332865715, "clip_ratio/region_mean": 0.04087404382880777, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 89.0, "completions/mean_terminated_length": 89.0, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.21917428076267242, "epoch": 0.09852592681849219, "frac_reward_zero_std": 0.0, "grad_norm": 4.428008079528809, "learning_rate": 2.5696969696969697e-06, "loss": -0.0104, "num_tokens": 5553819.0, "reward": 0.9670881032943726, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9919699430465698, "reward_meter_std": 0.005432384088635445, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06924089044332504, "reward_total_composite_mean": 0.9670881032943726, "reward_total_composite_std": 0.06924090534448624, "reward_total_mean": 0.9670881032943726, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9919699430465698, "rewards/meter/std": 0.005432384088635445, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9670881032943726, "rewards/total_composite/std": 0.06924090534448624, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0001150369644165, "sampling/importance_sampling_ratio/min": 0.35889124870300293, "sampling/sampling_logp_difference/max": 1.074939250946045, "sampling/sampling_logp_difference/mean": 0.031983181834220886, "step": 2453 }, { "clip_ratio/high_max": 0.0030487803742289543, "clip_ratio/high_mean": 0.0030487803742289543, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0030487803742289543, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 40.125, "completions/mean_terminated_length": 40.125, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.11128469556570053, "epoch": 0.09856609230027714, "frac_reward_zero_std": 0.0, "grad_norm": 4.521707534790039, "learning_rate": 2.566666666666667e-06, "loss": 0.0216, "num_tokens": 5555436.0, "reward": 0.9973416328430176, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973416328430176, "reward_meter_std": 0.0012014260282739997, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0012014324311167002, "reward_total_composite_mean": 0.9973416328430176, "reward_total_composite_std": 0.0012014260282739997, "reward_total_mean": 0.9973416328430176, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973416328430176, "rewards/meter/std": 0.0012014260282739997, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973416328430176, "rewards/total_composite/std": 0.0012014260282739997, "sampling/importance_sampling_ratio/max": 1.7687947750091553, "sampling/importance_sampling_ratio/mean": 0.9969314932823181, "sampling/importance_sampling_ratio/min": 0.2398810088634491, "sampling/sampling_logp_difference/max": 1.4276123046875, "sampling/sampling_logp_difference/mean": 0.017517054453492165, "step": 2454 }, { "clip_ratio/high_max": 0.007464349502697587, "clip_ratio/high_mean": 0.007464349502697587, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/region_mean": 0.009480478474870324, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 36.875, "completions/mean_terminated_length": 36.875, "completions/min_length": 33.0, "completions/min_terminated_length": 33.0, "entropy": 0.11671794764697552, "epoch": 0.0986062577820621, "frac_reward_zero_std": 0.0, "grad_norm": 2.2290995121002197, "learning_rate": 2.5636363636363638e-06, "loss": 0.2385, "num_tokens": 5556947.0, "reward": 0.8681668043136597, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.3535533845424652, "reward_meter_mean": 0.9920687675476074, "reward_meter_std": 0.00039521773578599095, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35079240798950195, "reward_total_composite_mean": 0.8681668043136597, "reward_total_composite_std": 0.35079243779182434, "reward_total_mean": 0.8681668043136597, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.3535533845424652, "rewards/meter/mean": 0.9920687675476074, "rewards/meter/std": 0.00039521773578599095, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8681668043136597, "rewards/total_composite/std": 0.35079243779182434, "sampling/importance_sampling_ratio/max": 1.3825856447219849, "sampling/importance_sampling_ratio/mean": 1.0059282779693604, "sampling/importance_sampling_ratio/min": 0.28730595111846924, "sampling/sampling_logp_difference/max": 1.2472076416015625, "sampling/sampling_logp_difference/mean": 0.016091084107756615, "step": 2455 }, { "clip_ratio/high_max": 0.007773919845931232, "clip_ratio/high_mean": 0.007773919845931232, "clip_ratio/low_mean": 0.003088302561081946, "clip_ratio/low_min": 0.003088302561081946, "clip_ratio/region_mean": 0.010862222407013178, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 80.5, "completions/mean_terminated_length": 80.5, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.22475284524261951, "epoch": 0.09864642326384705, "frac_reward_zero_std": 0.0, "grad_norm": 3.4946515560150146, "learning_rate": 2.5606060606060606e-06, "loss": 0.0052, "num_tokens": 5558871.0, "reward": 0.9981932044029236, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981932044029236, "reward_meter_std": 0.0004672010545618832, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004671814094763249, "reward_total_composite_mean": 0.9981932044029236, "reward_total_composite_std": 0.0004672010545618832, "reward_total_mean": 0.9981932044029236, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981932044029236, "rewards/meter/std": 0.0004672010545618832, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981932044029236, "rewards/total_composite/std": 0.0004672010545618832, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0071548223495483, "sampling/importance_sampling_ratio/min": 0.3567933142185211, "sampling/sampling_logp_difference/max": 1.0305986404418945, "sampling/sampling_logp_difference/mean": 0.02568862773478031, "step": 2456 }, { "clip_ratio/high_max": 0.00900993263348937, "clip_ratio/high_mean": 0.00900993263348937, "clip_ratio/low_mean": 0.01427258289186284, "clip_ratio/low_min": 0.01427258289186284, "clip_ratio/region_mean": 0.02328251552535221, "completions/clipped_ratio": 0.0, "completions/max_length": 190.0, "completions/max_terminated_length": 190.0, "completions/mean_length": 183.625, "completions/mean_terminated_length": 183.625, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.2139601055532694, "epoch": 0.098686588745632, "frac_reward_zero_std": 0.0, "grad_norm": 2.7975685596466064, "learning_rate": 2.5575757575757574e-06, "loss": 0.0114, "num_tokens": 5561940.0, "reward": 0.9250072836875916, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9926738142967224, "reward_meter_std": 0.000535113038495183, "reward_repeat_penalty_mean": 0.9318182468414307, "reward_repeat_penalty_std": 0.04208271950483322, "reward_std": 0.042194657027721405, "reward_total_composite_mean": 0.9250072836875916, "reward_total_composite_std": 0.042194660753011703, "reward_total_mean": 0.9250072836875916, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9926738142967224, "rewards/meter/std": 0.000535113038495183, "rewards/repeat_penalty/mean": 0.9318182468414307, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.9250072836875916, "rewards/total_composite/std": 0.042194660753011703, "sampling/importance_sampling_ratio/max": 1.8111183643341064, "sampling/importance_sampling_ratio/mean": 0.9990267157554626, "sampling/importance_sampling_ratio/min": 0.045105695724487305, "sampling/sampling_logp_difference/max": 3.0987467765808105, "sampling/sampling_logp_difference/mean": 0.03491319715976715, "step": 2457 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.005484737688675523, "epoch": 0.09872675422741696, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.5545454545454547e-06, "loss": 0.0, "num_tokens": 5563996.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0292819738388062, "sampling/importance_sampling_ratio/mean": 1.0000568628311157, "sampling/importance_sampling_ratio/min": 0.9144588112831116, "sampling/sampling_logp_difference/max": 0.08942283689975739, "sampling/sampling_logp_difference/mean": 0.00065056630410254, "step": 2458 }, { "clip_ratio/high_max": 0.03573530330322683, "clip_ratio/high_mean": 0.03573530330322683, "clip_ratio/low_mean": 0.009364139288663864, "clip_ratio/low_min": 0.009364139288663864, "clip_ratio/region_mean": 0.04509944259189069, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 164.0, "completions/mean_terminated_length": 164.0, "completions/min_length": 143.0, "completions/min_terminated_length": 143.0, "entropy": 0.37010542675852776, "epoch": 0.09876691970920191, "frac_reward_zero_std": 0.0, "grad_norm": 6.4525146484375, "learning_rate": 2.5515151515151515e-06, "loss": -0.0206, "num_tokens": 5566684.0, "reward": 0.9674677848815918, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948143362998962, "reward_meter_std": 0.010059010237455368, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.057435326278209686, "reward_total_composite_mean": 0.9674677848815918, "reward_total_composite_std": 0.05743533372879028, "reward_total_mean": 0.9674677848815918, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948143362998962, "rewards/meter/std": 0.010059010237455368, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9674677848815918, "rewards/total_composite/std": 0.05743533372879028, "sampling/importance_sampling_ratio/max": 1.882559061050415, "sampling/importance_sampling_ratio/mean": 1.0046672821044922, "sampling/importance_sampling_ratio/min": 0.1767299622297287, "sampling/sampling_logp_difference/max": 1.7331323623657227, "sampling/sampling_logp_difference/mean": 0.054715219885110855, "step": 2459 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0032962941040750593, "epoch": 0.09880708519098687, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.5484848484848484e-06, "loss": 0.0, "num_tokens": 5568492.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0136330127716064, "sampling/importance_sampling_ratio/mean": 1.0002198219299316, "sampling/importance_sampling_ratio/min": 0.9955151677131653, "sampling/sampling_logp_difference/max": 0.013540960848331451, "sampling/sampling_logp_difference/mean": 0.0002486900775693357, "step": 2460 }, { "clip_ratio/high_max": 0.022989682853221893, "clip_ratio/high_mean": 0.022989682853221893, "clip_ratio/low_mean": 0.008652225020341575, "clip_ratio/low_min": 0.008652225020341575, "clip_ratio/region_mean": 0.03164190787356347, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 404.5, "completions/mean_terminated_length": 404.5, "completions/min_length": 378.0, "completions/min_terminated_length": 378.0, "entropy": 0.35317912697792053, "epoch": 0.09884725067277182, "frac_reward_zero_std": 0.0, "grad_norm": 1.6912506818771362, "learning_rate": 2.5454545454545456e-06, "loss": -0.023, "num_tokens": 5573680.0, "reward": 0.7565553188323975, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7980769276618958, "reward_count_adherence_std": 0.05723259598016739, "reward_meter_mean": 0.9983578324317932, "reward_meter_std": 0.0009692087187431753, "reward_repeat_penalty_mean": 0.9470028877258301, "reward_repeat_penalty_std": 0.06291526556015015, "reward_std": 0.09492312371730804, "reward_total_composite_mean": 0.7565553188323975, "reward_total_composite_std": 0.09492313861846924, "reward_total_mean": 0.7565553188323975, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7980769276618958, "rewards/count_adherence/std": 0.05723259598016739, "rewards/meter/mean": 0.9983578324317932, "rewards/meter/std": 0.0009692087187431753, "rewards/repeat_penalty/mean": 0.9470028877258301, "rewards/repeat_penalty/std": 0.06291526556015015, "rewards/total_composite/mean": 0.7565553188323975, "rewards/total_composite/std": 0.09492313861846924, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0055227279663086, "sampling/importance_sampling_ratio/min": 0.17059072852134705, "sampling/sampling_logp_difference/max": 1.7684879302978516, "sampling/sampling_logp_difference/mean": 0.042006395757198334, "step": 2461 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.005597014795057476, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.017152427230030298, "epoch": 0.09888741615455678, "frac_reward_zero_std": 0.0, "grad_norm": 0.51884925365448, "learning_rate": 2.542424242424243e-06, "loss": 0.0004, "num_tokens": 5575497.0, "reward": 0.9981344938278198, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981344938278198, "reward_meter_std": 2.5179730073432438e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.518050132493954e-05, "reward_total_composite_mean": 0.9981344938278198, "reward_total_composite_std": 2.5179730073432438e-05, "reward_total_mean": 0.9981344938278198, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981344938278198, "rewards/meter/std": 2.5179730073432438e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981344938278198, "rewards/total_composite/std": 2.5179730073432438e-05, "sampling/importance_sampling_ratio/max": 1.997531771659851, "sampling/importance_sampling_ratio/mean": 0.9990748167037964, "sampling/importance_sampling_ratio/min": 0.37477174401283264, "sampling/sampling_logp_difference/max": 0.981438159942627, "sampling/sampling_logp_difference/mean": 0.007205577101558447, "step": 2462 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/region_mean": 0.0012755101779475808, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.027404201216995716, "epoch": 0.09892758163634173, "frac_reward_zero_std": 0.0, "grad_norm": 0.14106905460357666, "learning_rate": 2.5393939393939397e-06, "loss": 0.0004, "num_tokens": 5577697.0, "reward": 0.9980115294456482, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980115294456482, "reward_meter_std": 1.5354547940660268e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.5354515198851004e-05, "reward_total_composite_mean": 0.9980115294456482, "reward_total_composite_std": 1.5354547940660268e-05, "reward_total_mean": 0.9980115294456482, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980115294456482, "rewards/meter/std": 1.5354547940660268e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980115294456482, "rewards/total_composite/std": 1.5354547940660268e-05, "sampling/importance_sampling_ratio/max": 1.2301760911941528, "sampling/importance_sampling_ratio/mean": 1.0014915466308594, "sampling/importance_sampling_ratio/min": 0.6574686169624329, "sampling/sampling_logp_difference/max": 0.4193582534790039, "sampling/sampling_logp_difference/mean": 0.0028446686919778585, "step": 2463 }, { "clip_ratio/high_max": 0.004032257944345474, "clip_ratio/high_mean": 0.004032257944345474, "clip_ratio/low_mean": 0.008408705238252878, "clip_ratio/low_min": 0.008408705238252878, "clip_ratio/region_mean": 0.012440963182598352, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.09446914121508598, "epoch": 0.09896774711812668, "frac_reward_zero_std": 0.0, "grad_norm": 2.84199595451355, "learning_rate": 2.536363636363637e-06, "loss": -0.0148, "num_tokens": 5579505.0, "reward": 0.9931963086128235, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9931963086128235, "reward_meter_std": 0.00066983891883865, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006698234938085079, "reward_total_composite_mean": 0.9931963086128235, "reward_total_composite_std": 0.00066983891883865, "reward_total_mean": 0.9931963086128235, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9931963086128235, "rewards/meter/std": 0.00066983891883865, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9931963086128235, "rewards/total_composite/std": 0.00066983891883865, "sampling/importance_sampling_ratio/max": 1.462441325187683, "sampling/importance_sampling_ratio/mean": 1.0031862258911133, "sampling/importance_sampling_ratio/min": 0.4533645212650299, "sampling/sampling_logp_difference/max": 0.7910587787628174, "sampling/sampling_logp_difference/mean": 0.013205956667661667, "step": 2464 }, { "clip_ratio/high_max": 0.028696169378235936, "clip_ratio/high_mean": 0.028696169378235936, "clip_ratio/low_mean": 0.006513258791528642, "clip_ratio/low_min": 0.006513258791528642, "clip_ratio/region_mean": 0.03520942816976458, "completions/clipped_ratio": 0.0, "completions/max_length": 279.0, "completions/max_terminated_length": 279.0, "completions/mean_length": 271.125, "completions/mean_terminated_length": 271.125, "completions/min_length": 264.0, "completions/min_terminated_length": 264.0, "entropy": 0.39818970672786236, "epoch": 0.09900791259991164, "frac_reward_zero_std": 0.0, "grad_norm": 1.982505440711975, "learning_rate": 2.5333333333333338e-06, "loss": 0.0112, "num_tokens": 5583250.0, "reward": 0.969672441482544, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998477578163147, "reward_meter_std": 0.0006886826595291495, "reward_repeat_penalty_mean": 0.9711538553237915, "reward_repeat_penalty_std": 0.05723259598016739, "reward_std": 0.05711015686392784, "reward_total_composite_mean": 0.969672441482544, "reward_total_composite_std": 0.057110171765089035, "reward_total_mean": 0.969672441482544, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998477578163147, "rewards/meter/std": 0.0006886826595291495, "rewards/repeat_penalty/mean": 0.9711538553237915, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.969672441482544, "rewards/total_composite/std": 0.057110171765089035, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0076007843017578, "sampling/importance_sampling_ratio/min": 0.14942169189453125, "sampling/sampling_logp_difference/max": 1.9009828567504883, "sampling/sampling_logp_difference/mean": 0.043706730008125305, "step": 2465 }, { "clip_ratio/high_max": 0.004401408368721604, "clip_ratio/high_mean": 0.004401408368721604, "clip_ratio/low_mean": 0.010514133609831333, "clip_ratio/low_min": 0.010514133609831333, "clip_ratio/region_mean": 0.014915541978552938, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 142.5, "completions/mean_terminated_length": 142.5, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.095588818192482, "epoch": 0.09904807808169659, "frac_reward_zero_std": 0.0, "grad_norm": 1.754108190536499, "learning_rate": 2.5303030303030306e-06, "loss": 0.0039, "num_tokens": 5585974.0, "reward": 0.9098852872848511, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990931153297424, "reward_meter_std": 0.00018217807519249618, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07382424175739288, "reward_total_composite_mean": 0.9098852872848511, "reward_total_composite_std": 0.07382422685623169, "reward_total_mean": 0.9098852872848511, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990931153297424, "rewards/meter/std": 0.00018217807519249618, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9098852872848511, "rewards/total_composite/std": 0.07382422685623169, "sampling/importance_sampling_ratio/max": 1.6116867065429688, "sampling/importance_sampling_ratio/mean": 1.005768060684204, "sampling/importance_sampling_ratio/min": 0.42555540800094604, "sampling/sampling_logp_difference/max": 0.8543601036071777, "sampling/sampling_logp_difference/mean": 0.014981756918132305, "step": 2466 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/region_mean": 0.0015625000232830644, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.013048171065747738, "epoch": 0.09908824356348155, "frac_reward_zero_std": 0.0, "grad_norm": 0.4351802468299866, "learning_rate": 2.5272727272727274e-06, "loss": 0.0003, "num_tokens": 5587830.0, "reward": 0.6950865983963013, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7115719318389893, "reward_meter_std": 0.021074457094073296, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06770216673612595, "reward_total_composite_mean": 0.6950865983963013, "reward_total_composite_std": 0.06770216673612595, "reward_total_mean": 0.6950865983963013, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7115719318389893, "rewards/meter/std": 0.021074457094073296, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.6950865983963013, "rewards/total_composite/std": 0.06770216673612595, "sampling/importance_sampling_ratio/max": 1.3253552913665771, "sampling/importance_sampling_ratio/mean": 1.0006452798843384, "sampling/importance_sampling_ratio/min": 0.8967838287353516, "sampling/sampling_logp_difference/max": 0.2816805839538574, "sampling/sampling_logp_difference/mean": 0.0018030240898951888, "step": 2467 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0013736404580413364, "epoch": 0.0991284090452665, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.5242424242424247e-06, "loss": 0.0, "num_tokens": 5589294.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.003098964691162, "sampling/importance_sampling_ratio/mean": 1.0000958442687988, "sampling/importance_sampling_ratio/min": 0.9973318576812744, "sampling/sampling_logp_difference/max": 0.0030941502191126347, "sampling/sampling_logp_difference/mean": 0.00015046881162561476, "step": 2468 }, { "clip_ratio/high_max": 0.023276447434909642, "clip_ratio/high_mean": 0.023276447434909642, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.023276447434909642, "completions/clipped_ratio": 0.0, "completions/max_length": 95.0, "completions/max_terminated_length": 95.0, "completions/mean_length": 90.375, "completions/mean_terminated_length": 90.375, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.08473077462986112, "epoch": 0.09916857452705145, "frac_reward_zero_std": 0.0, "grad_norm": 4.22209358215332, "learning_rate": 2.5212121212121215e-06, "loss": -0.0232, "num_tokens": 5591249.0, "reward": 0.9691168665885925, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939854145050049, "reward_meter_std": 0.0009144411887973547, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07003801316022873, "reward_total_composite_mean": 0.9691168665885925, "reward_total_composite_std": 0.07003802061080933, "reward_total_mean": 0.9691168665885925, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939854145050049, "rewards/meter/std": 0.0009144411887973547, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9691168665885925, "rewards/total_composite/std": 0.07003802061080933, "sampling/importance_sampling_ratio/max": 1.7310431003570557, "sampling/importance_sampling_ratio/mean": 0.9984654784202576, "sampling/importance_sampling_ratio/min": 0.1446296125650406, "sampling/sampling_logp_difference/max": 1.9335792064666748, "sampling/sampling_logp_difference/mean": 0.022249562665820122, "step": 2469 }, { "clip_ratio/high_max": 0.03505220194347203, "clip_ratio/high_mean": 0.03505220194347203, "clip_ratio/low_mean": 0.010085079353302717, "clip_ratio/low_min": 0.010085079353302717, "clip_ratio/region_mean": 0.045137281296774745, "completions/clipped_ratio": 0.0, "completions/max_length": 168.0, "completions/max_terminated_length": 168.0, "completions/mean_length": 163.75, "completions/mean_terminated_length": 163.75, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.4189731851220131, "epoch": 0.09920874000883641, "frac_reward_zero_std": 0.0, "grad_norm": 2.6985630989074707, "learning_rate": 2.5181818181818184e-06, "loss": -0.0038, "num_tokens": 5594095.0, "reward": 0.9708901643753052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998643159866333, "reward_meter_std": 0.0008321683271788061, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05110771581530571, "reward_total_composite_mean": 0.9708901643753052, "reward_total_composite_std": 0.0511077418923378, "reward_total_mean": 0.9708901643753052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998643159866333, "rewards/meter/std": 0.0008321683271788061, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9708901643753052, "rewards/total_composite/std": 0.0511077418923378, "sampling/importance_sampling_ratio/max": 1.9667812585830688, "sampling/importance_sampling_ratio/mean": 1.012289047241211, "sampling/importance_sampling_ratio/min": 0.24335075914859772, "sampling/sampling_logp_difference/max": 1.4132513999938965, "sampling/sampling_logp_difference/mean": 0.049378104507923126, "step": 2470 }, { "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/low_mean": 0.010714285774156451, "clip_ratio/low_min": 0.010714285774156451, "clip_ratio/region_mean": 0.014285714365541935, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.1270341333001852, "epoch": 0.09924890549062136, "frac_reward_zero_std": 0.0, "grad_norm": 1.5141384601593018, "learning_rate": 2.5151515151515156e-06, "loss": -0.0009, "num_tokens": 5595647.0, "reward": 0.9994994401931763, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994994401931763, "reward_meter_std": 5.432292527984828e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.4318112233886495e-05, "reward_total_composite_mean": 0.9994994401931763, "reward_total_composite_std": 5.432292527984828e-05, "reward_total_mean": 0.9994994401931763, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994994401931763, "rewards/meter/std": 5.432292527984828e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994994401931763, "rewards/total_composite/std": 5.432292527984828e-05, "sampling/importance_sampling_ratio/max": 1.4859408140182495, "sampling/importance_sampling_ratio/mean": 1.0052924156188965, "sampling/importance_sampling_ratio/min": 0.33367058634757996, "sampling/sampling_logp_difference/max": 1.097601056098938, "sampling/sampling_logp_difference/mean": 0.019814442843198776, "step": 2471 }, { "clip_ratio/high_max": 0.002358490601181984, "clip_ratio/high_mean": 0.002358490601181984, "clip_ratio/low_mean": 0.006092437193728983, "clip_ratio/low_min": 0.006092437193728983, "clip_ratio/region_mean": 0.008450927794910967, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 105.25, "completions/mean_terminated_length": 105.25, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.025531375431455672, "epoch": 0.09928907097240632, "frac_reward_zero_std": 0.0, "grad_norm": 3.565403699874878, "learning_rate": 2.5121212121212125e-06, "loss": -0.0036, "num_tokens": 5598009.0, "reward": 0.38120073080062866, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.45625758171081543, "reward_meter_std": 0.3081152141094208, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.2558993995189667, "reward_total_composite_mean": 0.38120073080062866, "reward_total_composite_std": 0.2558993995189667, "reward_total_mean": 0.38120073080062866, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.45625758171081543, "rewards/meter/std": 0.3081152141094208, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.38120073080062866, "rewards/total_composite/std": 0.2558993995189667, "sampling/importance_sampling_ratio/max": 1.5240379571914673, "sampling/importance_sampling_ratio/mean": 1.0014318227767944, "sampling/importance_sampling_ratio/min": 0.5921472907066345, "sampling/sampling_logp_difference/max": 0.5239999294281006, "sampling/sampling_logp_difference/mean": 0.005791721399873495, "step": 2472 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0028824398177675903, "epoch": 0.09932923645419127, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.5090909090909093e-06, "loss": 0.0, "num_tokens": 5599793.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.010071873664856, "sampling/importance_sampling_ratio/mean": 1.0000567436218262, "sampling/importance_sampling_ratio/min": 0.9836629033088684, "sampling/sampling_logp_difference/max": 0.016472017392516136, "sampling/sampling_logp_difference/mean": 0.0002734074951149523, "step": 2473 }, { "clip_ratio/high_max": 0.003954203391913325, "clip_ratio/high_mean": 0.003954203391913325, "clip_ratio/low_mean": 0.015850404044613242, "clip_ratio/low_min": 0.015850404044613242, "clip_ratio/region_mean": 0.019804607436526567, "completions/clipped_ratio": 0.0, "completions/max_length": 223.0, "completions/max_terminated_length": 223.0, "completions/mean_length": 221.0, "completions/mean_terminated_length": 221.0, "completions/min_length": 218.0, "completions/min_terminated_length": 218.0, "entropy": 0.079356012865901, "epoch": 0.09936940193597622, "frac_reward_zero_std": 0.0, "grad_norm": 1.5029007196426392, "learning_rate": 2.506060606060606e-06, "loss": 0.0038, "num_tokens": 5603217.0, "reward": 0.7016157507896423, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9701865911483765, "reward_meter_std": 0.02245534211397171, "reward_repeat_penalty_mean": 0.7232142686843872, "reward_repeat_penalty_std": 0.05960877984762192, "reward_std": 0.060039594769477844, "reward_total_composite_mean": 0.7016157507896423, "reward_total_composite_std": 0.06003959849476814, "reward_total_mean": 0.7016157507896423, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9701865911483765, "rewards/meter/std": 0.02245534211397171, "rewards/repeat_penalty/mean": 0.7232142686843872, "rewards/repeat_penalty/std": 0.05960877984762192, "rewards/total_composite/mean": 0.7016157507896423, "rewards/total_composite/std": 0.06003959849476814, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9988224506378174, "sampling/importance_sampling_ratio/min": 0.0058447024784982204, "sampling/sampling_logp_difference/max": 5.142219543457031, "sampling/sampling_logp_difference/mean": 0.02541784755885601, "step": 2474 }, { "clip_ratio/high_max": 0.010605149669572711, "clip_ratio/high_mean": 0.010605149669572711, "clip_ratio/low_mean": 0.006118385819718242, "clip_ratio/low_min": 0.006118385819718242, "clip_ratio/region_mean": 0.016723535489290953, "completions/clipped_ratio": 0.0, "completions/max_length": 248.0, "completions/max_terminated_length": 248.0, "completions/mean_length": 246.375, "completions/mean_terminated_length": 246.375, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.14181741513311863, "epoch": 0.09940956741776118, "frac_reward_zero_std": 0.0, "grad_norm": 1.4216914176940918, "learning_rate": 2.5030303030303034e-06, "loss": 0.0005, "num_tokens": 5606940.0, "reward": 0.7197900414466858, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980918169021606, "reward_meter_std": 0.00017555728845763952, "reward_repeat_penalty_mean": 0.7211538553237915, "reward_repeat_penalty_std": 0.1158415824174881, "reward_std": 0.11570560187101364, "reward_total_composite_mean": 0.7197900414466858, "reward_total_composite_std": 0.11570559442043304, "reward_total_mean": 0.7197900414466858, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980918169021606, "rewards/meter/std": 0.00017555728845763952, "rewards/repeat_penalty/mean": 0.7211538553237915, "rewards/repeat_penalty/std": 0.1158415824174881, "rewards/total_composite/mean": 0.7197900414466858, "rewards/total_composite/std": 0.11570559442043304, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004142165184021, "sampling/importance_sampling_ratio/min": 0.35245901346206665, "sampling/sampling_logp_difference/max": 1.042820930480957, "sampling/sampling_logp_difference/mean": 0.020571939647197723, "step": 2475 }, { "clip_ratio/high_max": 0.02171556802932173, "clip_ratio/high_mean": 0.02171556802932173, "clip_ratio/low_mean": 0.014760755351744592, "clip_ratio/low_min": 0.014760755351744592, "clip_ratio/region_mean": 0.03647632338106632, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.25, "completions/mean_terminated_length": 68.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.3049812354147434, "epoch": 0.09944973289954613, "frac_reward_zero_std": 0.0, "grad_norm": 3.1809544563293457, "learning_rate": 2.5e-06, "loss": -0.003, "num_tokens": 5608790.0, "reward": 0.9992682933807373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992682933807373, "reward_meter_std": 0.00015381410776171833, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00015381253615487367, "reward_total_composite_mean": 0.9992682933807373, "reward_total_composite_std": 0.00015381410776171833, "reward_total_mean": 0.9992682933807373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992682933807373, "rewards/meter/std": 0.00015381410776171833, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992682933807373, "rewards/total_composite/std": 0.00015381410776171833, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0070955753326416, "sampling/importance_sampling_ratio/min": 0.3855144679546356, "sampling/sampling_logp_difference/max": 0.9773459434509277, "sampling/sampling_logp_difference/mean": 0.03672777861356735, "step": 2476 }, { "clip_ratio/high_max": 0.02343087922781706, "clip_ratio/high_mean": 0.02343087922781706, "clip_ratio/low_mean": 0.015457271714694798, "clip_ratio/low_min": 0.015457271714694798, "clip_ratio/region_mean": 0.03888815094251186, "completions/clipped_ratio": 0.0, "completions/max_length": 372.0, "completions/max_terminated_length": 372.0, "completions/mean_length": 361.125, "completions/mean_terminated_length": 361.125, "completions/min_length": 339.0, "completions/min_terminated_length": 339.0, "entropy": 0.3918083682656288, "epoch": 0.09948989838133108, "frac_reward_zero_std": 0.0, "grad_norm": 2.115460157394409, "learning_rate": 2.496969696969697e-06, "loss": 0.001, "num_tokens": 5613495.0, "reward": 0.714102029800415, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7857142686843872, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980402588844299, "reward_meter_std": 0.00200492306612432, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.06458108127117157, "reward_std": 0.049699652940034866, "reward_total_composite_mean": 0.714102029800415, "reward_total_composite_std": 0.04969966039061546, "reward_total_mean": 0.714102029800415, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7857142686843872, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980402588844299, "rewards/meter/std": 0.00200492306612432, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.06458108127117157, "rewards/total_composite/mean": 0.714102029800415, "rewards/total_composite/std": 0.04969966039061546, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0080512762069702, "sampling/importance_sampling_ratio/min": 0.14493077993392944, "sampling/sampling_logp_difference/max": 1.9314990043640137, "sampling/sampling_logp_difference/mean": 0.047193821519613266, "step": 2477 }, { "clip_ratio/high_max": 0.036573544377461076, "clip_ratio/high_mean": 0.036573544377461076, "clip_ratio/low_mean": 0.0020833334419876337, "clip_ratio/low_min": 0.0020833334419876337, "clip_ratio/region_mean": 0.03865687781944871, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 120.0, "completions/mean_terminated_length": 120.0, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.3629257455468178, "epoch": 0.09953006386311604, "frac_reward_zero_std": 0.0, "grad_norm": 6.063255786895752, "learning_rate": 2.4939393939393943e-06, "loss": 0.0048, "num_tokens": 5615759.0, "reward": 0.9968112111091614, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9968112111091614, "reward_meter_std": 0.004492159932851791, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004492142703384161, "reward_total_composite_mean": 0.9968112111091614, "reward_total_composite_std": 0.004492159932851791, "reward_total_mean": 0.9968112111091614, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9968112111091614, "rewards/meter/std": 0.004492159932851791, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9968112111091614, "rewards/total_composite/std": 0.004492159932851791, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0102128982543945, "sampling/importance_sampling_ratio/min": 0.42868536710739136, "sampling/sampling_logp_difference/max": 0.8470320701599121, "sampling/sampling_logp_difference/mean": 0.03631357103586197, "step": 2478 }, { "clip_ratio/high_max": 0.00933689041994512, "clip_ratio/high_mean": 0.00933689041994512, "clip_ratio/low_mean": 0.010923440335318446, "clip_ratio/low_min": 0.010923440335318446, "clip_ratio/region_mean": 0.020260330755263567, "completions/clipped_ratio": 0.0, "completions/max_length": 83.0, "completions/max_terminated_length": 83.0, "completions/mean_length": 80.375, "completions/mean_terminated_length": 80.375, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.26199326291680336, "epoch": 0.09957022934490099, "frac_reward_zero_std": 0.0, "grad_norm": 5.862101078033447, "learning_rate": 2.490909090909091e-06, "loss": 0.0172, "num_tokens": 5617602.0, "reward": 0.9985413551330566, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985413551330566, "reward_meter_std": 0.0006061770836822689, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006061761523596942, "reward_total_composite_mean": 0.9985413551330566, "reward_total_composite_std": 0.0006061770836822689, "reward_total_mean": 0.9985413551330566, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985413551330566, "rewards/meter/std": 0.0006061770836822689, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9985413551330566, "rewards/total_composite/std": 0.0006061770836822689, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0086257457733154, "sampling/importance_sampling_ratio/min": 0.2688840925693512, "sampling/sampling_logp_difference/max": 1.3134748935699463, "sampling/sampling_logp_difference/mean": 0.03767424076795578, "step": 2479 }, { "clip_ratio/high_max": 0.016681208508089185, "clip_ratio/high_mean": 0.016681208508089185, "clip_ratio/low_mean": 0.009811883908696473, "clip_ratio/low_min": 0.009811883908696473, "clip_ratio/region_mean": 0.026493092416785657, "completions/clipped_ratio": 0.0, "completions/max_length": 130.0, "completions/max_terminated_length": 130.0, "completions/mean_length": 127.125, "completions/mean_terminated_length": 127.125, "completions/min_length": 126.0, "completions/min_terminated_length": 126.0, "entropy": 0.25193316861987114, "epoch": 0.09961039482668595, "frac_reward_zero_std": 0.0, "grad_norm": 2.8270602226257324, "learning_rate": 2.487878787878788e-06, "loss": 0.01, "num_tokens": 5619995.0, "reward": 0.7446907758712769, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7682651281356812, "reward_meter_std": 0.1492764949798584, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.17325717210769653, "reward_total_composite_mean": 0.7446907758712769, "reward_total_composite_std": 0.17325718700885773, "reward_total_mean": 0.7446907758712769, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7682651281356812, "rewards/meter/std": 0.1492764949798584, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.7446907758712769, "rewards/total_composite/std": 0.17325718700885773, "sampling/importance_sampling_ratio/max": 1.9678964614868164, "sampling/importance_sampling_ratio/mean": 1.0069787502288818, "sampling/importance_sampling_ratio/min": 0.14564606547355652, "sampling/sampling_logp_difference/max": 1.9265758991241455, "sampling/sampling_logp_difference/mean": 0.03481648862361908, "step": 2480 }, { "clip_ratio/high_max": 0.02106643421575427, "clip_ratio/high_mean": 0.02106643421575427, "clip_ratio/low_mean": 0.009328358108177781, "clip_ratio/low_min": 0.009328358108177781, "clip_ratio/region_mean": 0.03039479232393205, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.625, "completions/mean_terminated_length": 65.625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.1946810930967331, "epoch": 0.0996505603084709, "frac_reward_zero_std": 0.0, "grad_norm": 5.6794843673706055, "learning_rate": 2.4848484848484848e-06, "loss": 0.0178, "num_tokens": 5621792.0, "reward": 0.9421236515045166, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9421236515045166, "reward_meter_std": 0.09346310049295425, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.09346310794353485, "reward_total_composite_mean": 0.9421236515045166, "reward_total_composite_std": 0.09346310049295425, "reward_total_mean": 0.9421236515045166, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9421236515045166, "rewards/meter/std": 0.09346310049295425, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9421236515045166, "rewards/total_composite/std": 0.09346310049295425, "sampling/importance_sampling_ratio/max": 1.7557963132858276, "sampling/importance_sampling_ratio/mean": 1.0088194608688354, "sampling/importance_sampling_ratio/min": 0.5547833442687988, "sampling/sampling_logp_difference/max": 0.5891776084899902, "sampling/sampling_logp_difference/mean": 0.02599693275988102, "step": 2481 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0015625000232830644, "clip_ratio/low_min": 0.0015625000232830644, "clip_ratio/region_mean": 0.0015625000232830644, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 60.5, "completions/mean_terminated_length": 60.5, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.007361570082139224, "epoch": 0.09969072579025585, "frac_reward_zero_std": 0.0, "grad_norm": 28.66371726989746, "learning_rate": 2.481818181818182e-06, "loss": 0.1809, "num_tokens": 5623580.0, "reward": 0.6806069612503052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.2314550280570984, "reward_meter_mean": 0.7704848051071167, "reward_meter_std": 0.03176296874880791, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.19818443059921265, "reward_total_composite_mean": 0.6806069612503052, "reward_total_composite_std": 0.19818444550037384, "reward_total_mean": 0.6806069612503052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.2314550280570984, "rewards/meter/mean": 0.7704848051071167, "rewards/meter/std": 0.03176296874880791, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6806069612503052, "rewards/total_composite/std": 0.19818444550037384, "sampling/importance_sampling_ratio/max": 1.083103060722351, "sampling/importance_sampling_ratio/mean": 1.0007379055023193, "sampling/importance_sampling_ratio/min": 0.9490934610366821, "sampling/sampling_logp_difference/max": 0.07983016967773438, "sampling/sampling_logp_difference/mean": 0.001127883093431592, "step": 2482 }, { "clip_ratio/high_max": 0.00828052987344563, "clip_ratio/high_mean": 0.00828052987344563, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/region_mean": 0.012312787817791104, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.25, "completions/mean_terminated_length": 61.25, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.049932122230529785, "epoch": 0.09973089127204081, "frac_reward_zero_std": 0.0, "grad_norm": 3.7192165851593018, "learning_rate": 2.478787878787879e-06, "loss": 0.0225, "num_tokens": 5625374.0, "reward": 0.9937635660171509, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9937635660171509, "reward_meter_std": 0.0006408250774256885, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006408333429135382, "reward_total_composite_mean": 0.9937635660171509, "reward_total_composite_std": 0.0006408250774256885, "reward_total_mean": 0.9937635660171509, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9937635660171509, "rewards/meter/std": 0.0006408250774256885, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9937635660171509, "rewards/total_composite/std": 0.0006408250774256885, "sampling/importance_sampling_ratio/max": 1.8052740097045898, "sampling/importance_sampling_ratio/mean": 0.998705267906189, "sampling/importance_sampling_ratio/min": 0.20806507766246796, "sampling/sampling_logp_difference/max": 1.5699043273925781, "sampling/sampling_logp_difference/mean": 0.015063888393342495, "step": 2483 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.002001831730012782, "epoch": 0.09977105675382576, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.475757575757576e-06, "loss": 0.0, "num_tokens": 5627126.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.00798761844635, "sampling/importance_sampling_ratio/mean": 1.0001825094223022, "sampling/importance_sampling_ratio/min": 0.9975246787071228, "sampling/sampling_logp_difference/max": 0.007955902256071568, "sampling/sampling_logp_difference/mean": 0.00020638658315874636, "step": 2484 }, { "clip_ratio/high_max": 0.021971354726701975, "clip_ratio/high_mean": 0.021971354726701975, "clip_ratio/low_mean": 0.01650261995382607, "clip_ratio/low_min": 0.01650261995382607, "clip_ratio/region_mean": 0.038473974680528045, "completions/clipped_ratio": 0.0, "completions/max_length": 393.0, "completions/max_terminated_length": 393.0, "completions/mean_length": 363.625, "completions/mean_terminated_length": 363.625, "completions/min_length": 343.0, "completions/min_terminated_length": 343.0, "entropy": 0.4474870078265667, "epoch": 0.09981122223561072, "frac_reward_zero_std": 0.0, "grad_norm": 1.8019484281539917, "learning_rate": 2.472727272727273e-06, "loss": -0.0278, "num_tokens": 5631659.0, "reward": 0.8964345455169678, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9249999523162842, "reward_count_adherence_std": 0.04629101976752281, "reward_meter_mean": 0.9972596764564514, "reward_meter_std": 0.004136989358812571, "reward_repeat_penalty_mean": 0.9709967374801636, "reward_repeat_penalty_std": 0.031024247407913208, "reward_std": 0.0664711743593216, "reward_total_composite_mean": 0.8964345455169678, "reward_total_composite_std": 0.0664711743593216, "reward_total_mean": 0.8964345455169678, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9249999523162842, "rewards/count_adherence/std": 0.04629101976752281, "rewards/meter/mean": 0.9972596764564514, "rewards/meter/std": 0.004136989358812571, "rewards/repeat_penalty/mean": 0.9709967374801636, "rewards/repeat_penalty/std": 0.031024247407913208, "rewards/total_composite/mean": 0.8964345455169678, "rewards/total_composite/std": 0.0664711743593216, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0112888813018799, "sampling/importance_sampling_ratio/min": 0.08302469551563263, "sampling/sampling_logp_difference/max": 2.488617181777954, "sampling/sampling_logp_difference/mean": 0.052424363791942596, "step": 2485 }, { "clip_ratio/high_max": 0.03322292352095246, "clip_ratio/high_mean": 0.03322292352095246, "clip_ratio/low_mean": 0.005464080721139908, "clip_ratio/low_min": 0.005464080721139908, "clip_ratio/region_mean": 0.03868700424209237, "completions/clipped_ratio": 0.0, "completions/max_length": 161.0, "completions/max_terminated_length": 161.0, "completions/mean_length": 158.5, "completions/mean_terminated_length": 158.5, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.45672084018588066, "epoch": 0.09985138771739567, "frac_reward_zero_std": 0.0, "grad_norm": 2.246661901473999, "learning_rate": 2.46969696969697e-06, "loss": 0.0133, "num_tokens": 5634271.0, "reward": 0.9984291791915894, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984291791915894, "reward_meter_std": 0.0007291806978173554, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007291696383617818, "reward_total_composite_mean": 0.9984291791915894, "reward_total_composite_std": 0.0007291806978173554, "reward_total_mean": 0.9984291791915894, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984291791915894, "rewards/meter/std": 0.0007291806978173554, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984291791915894, "rewards/total_composite/std": 0.0007291806978173554, "sampling/importance_sampling_ratio/max": 1.7283272743225098, "sampling/importance_sampling_ratio/mean": 1.0025769472122192, "sampling/importance_sampling_ratio/min": 0.223396435379982, "sampling/sampling_logp_difference/max": 1.498807430267334, "sampling/sampling_logp_difference/mean": 0.049818817526102066, "step": 2486 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0025851023383438587, "epoch": 0.09989155319918062, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.466666666666667e-06, "loss": 0.0, "num_tokens": 5636167.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0113502740859985, "sampling/importance_sampling_ratio/mean": 1.0001945495605469, "sampling/importance_sampling_ratio/min": 0.9940236806869507, "sampling/sampling_logp_difference/max": 0.011286328546702862, "sampling/sampling_logp_difference/mean": 0.00024826714070513844, "step": 2487 }, { "clip_ratio/high_max": 0.0038265305338427424, "clip_ratio/high_mean": 0.0038265305338427424, "clip_ratio/low_mean": 0.0013297871919348836, "clip_ratio/low_min": 0.0013297871919348836, "clip_ratio/region_mean": 0.005156317725777626, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.5, "completions/mean_terminated_length": 97.5, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.01773504843004048, "epoch": 0.09993171868096558, "frac_reward_zero_std": 0.0, "grad_norm": 0.9873250126838684, "learning_rate": 2.463636363636364e-06, "loss": -0.0124, "num_tokens": 5638299.0, "reward": 0.9964094758033752, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9964094758033752, "reward_meter_std": 0.004482356831431389, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00448233587667346, "reward_total_composite_mean": 0.9964094758033752, "reward_total_composite_std": 0.004482356831431389, "reward_total_mean": 0.9964094758033752, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9964094758033752, "rewards/meter/std": 0.004482356831431389, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964094758033752, "rewards/total_composite/std": 0.004482356831431389, "sampling/importance_sampling_ratio/max": 1.091353416442871, "sampling/importance_sampling_ratio/mean": 0.9996676445007324, "sampling/importance_sampling_ratio/min": 0.2779403328895569, "sampling/sampling_logp_difference/max": 1.280348777770996, "sampling/sampling_logp_difference/mean": 0.0035261192824691534, "step": 2488 }, { "clip_ratio/high_max": 0.03615045174956322, "clip_ratio/high_mean": 0.03615045174956322, "clip_ratio/low_mean": 0.05386740108951926, "clip_ratio/low_min": 0.05386740108951926, "clip_ratio/region_mean": 0.09001785283908248, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 23.75, "completions/mean_terminated_length": 23.75, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.5839281938970089, "epoch": 0.09997188416275053, "frac_reward_zero_std": 0.0, "grad_norm": 19.695940017700195, "learning_rate": 2.4606060606060607e-06, "loss": 0.0144, "num_tokens": 5639817.0, "reward": 0.9360402822494507, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9360402822494507, "reward_meter_std": 0.019930923357605934, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.019930915907025337, "reward_total_composite_mean": 0.9360402822494507, "reward_total_composite_std": 0.019930923357605934, "reward_total_mean": 0.9360402822494507, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9360402822494507, "rewards/meter/std": 0.019930923357605934, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9360402822494507, "rewards/total_composite/std": 0.019930923357605934, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0085498094558716, "sampling/importance_sampling_ratio/min": 0.3307119607925415, "sampling/sampling_logp_difference/max": 1.1065075397491455, "sampling/sampling_logp_difference/mean": 0.09426023066043854, "step": 2489 }, { "clip_ratio/high_max": 0.008156140334904194, "clip_ratio/high_mean": 0.008156140334904194, "clip_ratio/low_mean": 0.0022935778833925724, "clip_ratio/low_min": 0.0022935778833925724, "clip_ratio/region_mean": 0.010449718218296766, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 107.5, "completions/mean_terminated_length": 107.5, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.12307433690875769, "epoch": 0.10001204964453549, "frac_reward_zero_std": 0.0, "grad_norm": 1.8480373620986938, "learning_rate": 2.457575757575758e-06, "loss": 0.0052, "num_tokens": 5642037.0, "reward": 0.9727107882499695, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976269602775574, "reward_meter_std": 0.0004338659346103668, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.0708698257803917, "reward_total_composite_mean": 0.9727107882499695, "reward_total_composite_std": 0.0708698034286499, "reward_total_mean": 0.9727107882499695, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976269602775574, "rewards/meter/std": 0.0004338659346103668, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9727107882499695, "rewards/total_composite/std": 0.0708698034286499, "sampling/importance_sampling_ratio/max": 1.686613917350769, "sampling/importance_sampling_ratio/mean": 1.003526210784912, "sampling/importance_sampling_ratio/min": 0.09479877352714539, "sampling/sampling_logp_difference/max": 2.3559987545013428, "sampling/sampling_logp_difference/mean": 0.01844092458486557, "step": 2490 }, { "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/region_mean": 0.006048386916518211, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 62.0, "completions/mean_terminated_length": 62.0, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.040033137891441584, "epoch": 0.10005221512632044, "frac_reward_zero_std": 0.0, "grad_norm": 2.2775914669036865, "learning_rate": 2.454545454545455e-06, "loss": -0.0008, "num_tokens": 5643821.0, "reward": 0.9934873580932617, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9934873580932617, "reward_meter_std": 0.0007373938569799066, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000737398921046406, "reward_total_composite_mean": 0.9934873580932617, "reward_total_composite_std": 0.0007373938569799066, "reward_total_mean": 0.9934873580932617, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9934873580932617, "rewards/meter/std": 0.0007373938569799066, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934873580932617, "rewards/total_composite/std": 0.0007373938569799066, "sampling/importance_sampling_ratio/max": 1.28243088722229, "sampling/importance_sampling_ratio/mean": 1.001605749130249, "sampling/importance_sampling_ratio/min": 0.6251553297042847, "sampling/sampling_logp_difference/max": 0.4697551727294922, "sampling/sampling_logp_difference/mean": 0.007004985585808754, "step": 2491 }, { "clip_ratio/high_max": 0.01584376988466829, "clip_ratio/high_mean": 0.01584376988466829, "clip_ratio/low_mean": 0.0010080644860863686, "clip_ratio/low_min": 0.0010080644860863686, "clip_ratio/region_mean": 0.01685183437075466, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 125.625, "completions/mean_terminated_length": 125.625, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.10137755703181028, "epoch": 0.1000923806081054, "frac_reward_zero_std": 0.0, "grad_norm": 3.0206403732299805, "learning_rate": 2.4515151515151516e-06, "loss": -0.006, "num_tokens": 5646242.0, "reward": 0.9575905799865723, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9930568337440491, "reward_meter_std": 0.00019196225912310183, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06567277014255524, "reward_total_composite_mean": 0.9575905799865723, "reward_total_composite_std": 0.06567274034023285, "reward_total_mean": 0.9575905799865723, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9930568337440491, "rewards/meter/std": 0.00019196225912310183, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9575905799865723, "rewards/total_composite/std": 0.06567274034023285, "sampling/importance_sampling_ratio/max": 1.629332184791565, "sampling/importance_sampling_ratio/mean": 1.001326322555542, "sampling/importance_sampling_ratio/min": 0.09002399444580078, "sampling/sampling_logp_difference/max": 2.4076790809631348, "sampling/sampling_logp_difference/mean": 0.018952438607811928, "step": 2492 }, { "clip_ratio/high_max": 0.025025799637660384, "clip_ratio/high_mean": 0.025025799637660384, "clip_ratio/low_mean": 0.005151562392711639, "clip_ratio/low_min": 0.005151562392711639, "clip_ratio/region_mean": 0.030177362030372024, "completions/clipped_ratio": 0.0, "completions/max_length": 172.0, "completions/max_terminated_length": 172.0, "completions/mean_length": 162.25, "completions/mean_terminated_length": 162.25, "completions/min_length": 154.0, "completions/min_terminated_length": 154.0, "entropy": 0.4163212776184082, "epoch": 0.10013254608989035, "frac_reward_zero_std": 0.0, "grad_norm": 3.180861234664917, "learning_rate": 2.4484848484848485e-06, "loss": 0.0227, "num_tokens": 5649068.0, "reward": 0.881544828414917, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.904288649559021, "reward_meter_std": 0.09930223226547241, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.1262788623571396, "reward_total_composite_mean": 0.881544828414917, "reward_total_composite_std": 0.1262788623571396, "reward_total_mean": 0.881544828414917, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.904288649559021, "rewards/meter/std": 0.09930223226547241, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.881544828414917, "rewards/total_composite/std": 0.1262788623571396, "sampling/importance_sampling_ratio/max": 1.7914950847625732, "sampling/importance_sampling_ratio/mean": 1.0083729028701782, "sampling/importance_sampling_ratio/min": 0.19677433371543884, "sampling/sampling_logp_difference/max": 1.6256977319717407, "sampling/sampling_logp_difference/mean": 0.041033390909433365, "step": 2493 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.010702198022045195, "epoch": 0.1001727115716753, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.4454545454545457e-06, "loss": 0.0, "num_tokens": 5650820.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0194705724716187, "sampling/importance_sampling_ratio/mean": 1.000666856765747, "sampling/importance_sampling_ratio/min": 0.9093199372291565, "sampling/sampling_logp_difference/max": 0.09505826234817505, "sampling/sampling_logp_difference/mean": 0.0011890914756804705, "step": 2494 }, { "clip_ratio/high_max": 0.014545450918376446, "clip_ratio/high_mean": 0.014545450918376446, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.014545450918376446, "completions/clipped_ratio": 0.0, "completions/max_length": 96.0, "completions/max_terminated_length": 96.0, "completions/mean_length": 94.75, "completions/mean_terminated_length": 94.75, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.10162492096424103, "epoch": 0.10021287705346026, "frac_reward_zero_std": 0.0, "grad_norm": 1.8723995685577393, "learning_rate": 2.4424242424242426e-06, "loss": 0.0059, "num_tokens": 5652882.0, "reward": 0.9682368040084839, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9930741786956787, "reward_meter_std": 0.0006547096418216825, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07008406519889832, "reward_total_composite_mean": 0.9682368040084839, "reward_total_composite_std": 0.07008406519889832, "reward_total_mean": 0.9682368040084839, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9930741786956787, "rewards/meter/std": 0.0006547096418216825, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9682368040084839, "rewards/total_composite/std": 0.07008406519889832, "sampling/importance_sampling_ratio/max": 1.5056626796722412, "sampling/importance_sampling_ratio/mean": 1.0004006624221802, "sampling/importance_sampling_ratio/min": 0.24938809871673584, "sampling/sampling_logp_difference/max": 1.3887449502944946, "sampling/sampling_logp_difference/mean": 0.018359439447522163, "step": 2495 }, { "clip_ratio/high_max": 0.02076363516971469, "clip_ratio/high_mean": 0.02076363516971469, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/region_mean": 0.025393264833837748, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 105.5, "completions/mean_terminated_length": 105.5, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.23731265030801296, "epoch": 0.10025304253524521, "frac_reward_zero_std": 0.0, "grad_norm": 3.207080602645874, "learning_rate": 2.4393939393939394e-06, "loss": 0.0159, "num_tokens": 5655174.0, "reward": 0.9473263025283813, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972423315048218, "reward_meter_std": 0.0021226475946605206, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.0917830839753151, "reward_total_composite_mean": 0.9473263025283813, "reward_total_composite_std": 0.09178309142589569, "reward_total_mean": 0.9473263025283813, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972423315048218, "rewards/meter/std": 0.0021226475946605206, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9473263025283813, "rewards/total_composite/std": 0.09178309142589569, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0094503164291382, "sampling/importance_sampling_ratio/min": 0.2535927891731262, "sampling/sampling_logp_difference/max": 1.372025489807129, "sampling/sampling_logp_difference/mean": 0.03410215303301811, "step": 2496 }, { "clip_ratio/high_max": 0.05119223310612142, "clip_ratio/high_mean": 0.05119223310612142, "clip_ratio/low_mean": 0.004247572738677263, "clip_ratio/low_min": 0.004247572738677263, "clip_ratio/region_mean": 0.055439805844798684, "completions/clipped_ratio": 0.0, "completions/max_length": 206.0, "completions/max_terminated_length": 206.0, "completions/mean_length": 175.125, "completions/mean_terminated_length": 175.125, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.5811385102570057, "epoch": 0.10029320801703016, "frac_reward_zero_std": 0.0, "grad_norm": 5.552990913391113, "learning_rate": 2.4363636363636366e-06, "loss": 0.0681, "num_tokens": 5658127.0, "reward": 0.8185431361198425, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8541666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.9639856219291687, "reward_meter_std": 0.09431485831737518, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.035573359578847885, "reward_total_composite_mean": 0.8185431361198425, "reward_total_composite_std": 0.03557335212826729, "reward_total_mean": 0.8185431361198425, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8541666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.9639856219291687, "rewards/meter/std": 0.09431485831737518, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8185431361198425, "rewards/total_composite/std": 0.03557335212826729, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0070842504501343, "sampling/importance_sampling_ratio/min": 0.12097006291151047, "sampling/sampling_logp_difference/max": 2.1122121810913086, "sampling/sampling_logp_difference/mean": 0.0666884258389473, "step": 2497 }, { "clip_ratio/high_max": 0.01925921393558383, "clip_ratio/high_mean": 0.01925921393558383, "clip_ratio/low_mean": 0.0013020833721384406, "clip_ratio/low_min": 0.0013020833721384406, "clip_ratio/region_mean": 0.02056129730772227, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 97.125, "completions/mean_terminated_length": 97.125, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.30781341157853603, "epoch": 0.10033337349881512, "frac_reward_zero_std": 0.0, "grad_norm": 3.1228280067443848, "learning_rate": 2.4333333333333335e-06, "loss": -0.0013, "num_tokens": 5660216.0, "reward": 0.9413519501686096, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.991081714630127, "reward_meter_std": 0.004786557983607054, "reward_repeat_penalty_mean": 0.949999988079071, "reward_repeat_penalty_std": 0.09258200973272324, "reward_std": 0.09001474827528, "reward_total_composite_mean": 0.9413519501686096, "reward_total_composite_std": 0.09001474827528, "reward_total_mean": 0.9413519501686096, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.991081714630127, "rewards/meter/std": 0.004786557983607054, "rewards/repeat_penalty/mean": 0.949999988079071, "rewards/repeat_penalty/std": 0.09258200973272324, "rewards/total_composite/mean": 0.9413519501686096, "rewards/total_composite/std": 0.09001474827528, "sampling/importance_sampling_ratio/max": 1.695669412612915, "sampling/importance_sampling_ratio/mean": 1.0101373195648193, "sampling/importance_sampling_ratio/min": 0.43812206387519836, "sampling/sampling_logp_difference/max": 0.8252577781677246, "sampling/sampling_logp_difference/mean": 0.031947266310453415, "step": 2498 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.5, "completions/mean_terminated_length": 67.5, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.015344980405643582, "epoch": 0.10037353898060007, "frac_reward_zero_std": 0.0, "grad_norm": 3.2906627655029297, "learning_rate": 2.4303030303030307e-06, "loss": 0.008, "num_tokens": 5661956.0, "reward": 0.9980784058570862, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980784058570862, "reward_meter_std": 0.00020892193424515426, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020891590975224972, "reward_total_composite_mean": 0.9980784058570862, "reward_total_composite_std": 0.00020892193424515426, "reward_total_mean": 0.9980784058570862, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980784058570862, "rewards/meter/std": 0.00020892193424515426, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980784058570862, "rewards/total_composite/std": 0.00020892193424515426, "sampling/importance_sampling_ratio/max": 1.0832960605621338, "sampling/importance_sampling_ratio/mean": 0.9999598860740662, "sampling/importance_sampling_ratio/min": 0.35609135031700134, "sampling/sampling_logp_difference/max": 1.0325679779052734, "sampling/sampling_logp_difference/mean": 0.0033059667330235243, "step": 2499 }, { "clip_ratio/high_max": 0.004727728548459709, "clip_ratio/high_mean": 0.004727728548459709, "clip_ratio/low_mean": 0.011479852895718068, "clip_ratio/low_min": 0.011479852895718068, "clip_ratio/region_mean": 0.016207581444177777, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 131.375, "completions/mean_terminated_length": 131.375, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.05698791332542896, "epoch": 0.10041370446238503, "frac_reward_zero_std": 0.0, "grad_norm": 3.226564884185791, "learning_rate": 2.4272727272727276e-06, "loss": -0.0048, "num_tokens": 5664327.0, "reward": 0.9096584916114807, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998822808265686, "reward_meter_std": 0.0007065999088808894, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07408840954303741, "reward_total_composite_mean": 0.9096584916114807, "reward_total_composite_std": 0.07408842444419861, "reward_total_mean": 0.9096584916114807, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998822808265686, "rewards/meter/std": 0.0007065999088808894, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9096584916114807, "rewards/total_composite/std": 0.07408842444419861, "sampling/importance_sampling_ratio/max": 1.8086941242218018, "sampling/importance_sampling_ratio/mean": 0.9981679916381836, "sampling/importance_sampling_ratio/min": 0.17515164613723755, "sampling/sampling_logp_difference/max": 1.742103099822998, "sampling/sampling_logp_difference/mean": 0.01683160476386547, "step": 2500 }, { "epoch": 0.10041370446238503, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 405.3076923076923, "eval_completions/max_terminated_length": 405.3076923076923, "eval_completions/mean_length": 210.92307692307693, "eval_completions/mean_terminated_length": 210.92307692307693, "eval_completions/min_length": 62.38461538461539, "eval_completions/min_terminated_length": 62.38461538461539, "eval_entropy": 0.37157516181468964, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 5664327.0, "eval_reward": 0.6316862404346466, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_count_adherence_mean": 0.9494116031206571, "eval_reward_count_adherence_std": 0.07668151821081455, "eval_reward_meter_mean": 0.7411972742814285, "eval_reward_meter_std": 0.36808492687459177, "eval_reward_repeat_penalty_mean": 0.9134463484470661, "eval_reward_repeat_penalty_std": 0.10245848590364823, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6316862404346466, "eval_reward_total_composite_std": 0.3585601345850871, "eval_reward_total_mean": 0.6316862404346466, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/count_adherence/mean": 0.9494116031206571, "eval_rewards/count_adherence/std": 0.07668151821081455, "eval_rewards/meter/mean": 0.7411972742814285, "eval_rewards/meter/std": 0.36808492687459177, "eval_rewards/repeat_penalty/mean": 0.9134463484470661, "eval_rewards/repeat_penalty/std": 0.10245848590364823, "eval_rewards/total_composite/mean": 0.6316862404346466, "eval_rewards/total_composite/std": 0.3585601345850871, "eval_runtime": 75.4547, "eval_samples_per_second": 1.378, "eval_sampling/importance_sampling_ratio/max": 1.5564926862716675, "eval_sampling/importance_sampling_ratio/mean": 1.0096348799191988, "eval_sampling/importance_sampling_ratio/min": 0.31562885412803066, "eval_sampling/sampling_logp_difference/max": 1.1953271352327788, "eval_sampling/sampling_logp_difference/mean": 0.03223242281148067, "eval_steps_per_second": 0.172, "step": 2500 }, { "clip_ratio/high_max": 0.02857714961282909, "clip_ratio/high_mean": 0.02857714961282909, "clip_ratio/low_mean": 0.003677265834994614, "clip_ratio/low_min": 0.003677265834994614, "clip_ratio/region_mean": 0.0322544154478237, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.3523239102214575, "epoch": 0.10045386994416998, "frac_reward_zero_std": 0.0, "grad_norm": 4.538857460021973, "learning_rate": 2.4242424242424244e-06, "loss": 0.018, "num_tokens": 5666190.0, "reward": 0.9774847626686096, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9774847626686096, "reward_meter_std": 0.03591068461537361, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03591068089008331, "reward_total_composite_mean": 0.9774847626686096, "reward_total_composite_std": 0.03591068461537361, "reward_total_mean": 0.9774847626686096, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9774847626686096, "rewards/meter/std": 0.03591068461537361, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9774847626686096, "rewards/total_composite/std": 0.03591068461537361, "sampling/importance_sampling_ratio/max": 1.6371313333511353, "sampling/importance_sampling_ratio/mean": 1.0134010314941406, "sampling/importance_sampling_ratio/min": 0.44160670042037964, "sampling/sampling_logp_difference/max": 0.8173356056213379, "sampling/sampling_logp_difference/mean": 0.04367325082421303, "step": 2501 }, { "clip_ratio/high_max": 0.029540141811594367, "clip_ratio/high_mean": 0.029540141811594367, "clip_ratio/low_mean": 0.013037330703809857, "clip_ratio/low_min": 0.013037330703809857, "clip_ratio/region_mean": 0.042577472515404224, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.25, "completions/mean_terminated_length": 67.25, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.3921285644173622, "epoch": 0.10049403542595493, "frac_reward_zero_std": 0.0, "grad_norm": 4.3011322021484375, "learning_rate": 2.4212121212121216e-06, "loss": 0.0026, "num_tokens": 5668128.0, "reward": 0.9222630262374878, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9222630262374878, "reward_meter_std": 0.14857283234596252, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.14857284724712372, "reward_total_composite_mean": 0.9222630262374878, "reward_total_composite_std": 0.14857283234596252, "reward_total_mean": 0.9222630262374878, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9222630262374878, "rewards/meter/std": 0.14857283234596252, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9222630262374878, "rewards/total_composite/std": 0.14857283234596252, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0067012310028076, "sampling/importance_sampling_ratio/min": 0.4099763035774231, "sampling/sampling_logp_difference/max": 0.8916559219360352, "sampling/sampling_logp_difference/mean": 0.04170745983719826, "step": 2502 }, { "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/low_mean": 0.008732197224162519, "clip_ratio/low_min": 0.008732197224162519, "clip_ratio/region_mean": 0.012253323919139802, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.14498437847942114, "epoch": 0.10053420090773989, "frac_reward_zero_std": 0.0, "grad_norm": 4.316682815551758, "learning_rate": 2.4181818181818185e-06, "loss": 0.0043, "num_tokens": 5669907.0, "reward": 0.9978486895561218, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978486895561218, "reward_meter_std": 0.00048536990652792156, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00048536164104007185, "reward_total_composite_mean": 0.9978486895561218, "reward_total_composite_std": 0.00048536990652792156, "reward_total_mean": 0.9978486895561218, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978486895561218, "rewards/meter/std": 0.00048536990652792156, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978486895561218, "rewards/total_composite/std": 0.00048536990652792156, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0028010606765747, "sampling/importance_sampling_ratio/min": 0.37709715962409973, "sampling/sampling_logp_difference/max": 0.9752523899078369, "sampling/sampling_logp_difference/mean": 0.021549848839640617, "step": 2503 }, { "clip_ratio/high_max": 0.02037855191156268, "clip_ratio/high_mean": 0.02037855191156268, "clip_ratio/low_mean": 0.02393018058501184, "clip_ratio/low_min": 0.02393018058501184, "clip_ratio/region_mean": 0.04430873249657452, "completions/clipped_ratio": 0.0, "completions/max_length": 39.0, "completions/max_terminated_length": 39.0, "completions/mean_length": 36.75, "completions/mean_terminated_length": 36.75, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.32645551674067974, "epoch": 0.10057436638952484, "frac_reward_zero_std": 0.0, "grad_norm": 6.806858539581299, "learning_rate": 2.4151515151515153e-06, "loss": -0.0032, "num_tokens": 5671433.0, "reward": 0.9697108268737793, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9697108268737793, "reward_meter_std": 0.01697305217385292, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01697305217385292, "reward_total_composite_mean": 0.9697108268737793, "reward_total_composite_std": 0.01697305217385292, "reward_total_mean": 0.9697108268737793, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9697108268737793, "rewards/meter/std": 0.01697305217385292, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9697108268737793, "rewards/total_composite/std": 0.01697305217385292, "sampling/importance_sampling_ratio/max": 1.5999165773391724, "sampling/importance_sampling_ratio/mean": 1.0082935094833374, "sampling/importance_sampling_ratio/min": 0.5814112424850464, "sampling/sampling_logp_difference/max": 0.5422968864440918, "sampling/sampling_logp_difference/mean": 0.03803471848368645, "step": 2504 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018382353009656072, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.013536597951315343, "epoch": 0.1006145318713098, "frac_reward_zero_std": 0.0, "grad_norm": 0.20346412062644958, "learning_rate": 2.412121212121212e-06, "loss": -0.0003, "num_tokens": 5673402.0, "reward": 0.9981461763381958, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981461763381958, "reward_meter_std": 1.717484337859787e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.7183874660986476e-05, "reward_total_composite_mean": 0.9981461763381958, "reward_total_composite_std": 1.717484337859787e-05, "reward_total_mean": 0.9981461763381958, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981461763381958, "rewards/meter/std": 1.717484337859787e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981461763381958, "rewards/total_composite/std": 1.717484337859787e-05, "sampling/importance_sampling_ratio/max": 1.091694712638855, "sampling/importance_sampling_ratio/mean": 0.9994654655456543, "sampling/importance_sampling_ratio/min": 0.4915757179260254, "sampling/sampling_logp_difference/max": 0.710139274597168, "sampling/sampling_logp_difference/mean": 0.00340123544447124, "step": 2505 }, { "clip_ratio/high_max": 0.02579016692470759, "clip_ratio/high_mean": 0.02579016692470759, "clip_ratio/low_mean": 0.003846153849735856, "clip_ratio/low_min": 0.003846153849735856, "clip_ratio/region_mean": 0.029636320774443448, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.5, "completions/mean_terminated_length": 67.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.3341045305132866, "epoch": 0.10065469735309475, "frac_reward_zero_std": 0.0, "grad_norm": 6.047022342681885, "learning_rate": 2.4090909090909094e-06, "loss": -0.0096, "num_tokens": 5675142.0, "reward": 0.9976764917373657, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976764917373657, "reward_meter_std": 0.004249152261763811, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004249148536473513, "reward_total_composite_mean": 0.9976764917373657, "reward_total_composite_std": 0.004249152261763811, "reward_total_mean": 0.9976764917373657, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976764917373657, "rewards/meter/std": 0.004249152261763811, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976764917373657, "rewards/total_composite/std": 0.004249152261763811, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0117424726486206, "sampling/importance_sampling_ratio/min": 0.09082647413015366, "sampling/sampling_logp_difference/max": 2.3988044261932373, "sampling/sampling_logp_difference/mean": 0.04069520905613899, "step": 2506 }, { "clip_ratio/high_max": 0.008196721319109201, "clip_ratio/high_mean": 0.008196721319109201, "clip_ratio/low_mean": 0.015786210540682077, "clip_ratio/low_min": 0.015786210540682077, "clip_ratio/region_mean": 0.02398293185979128, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 62.625, "completions/mean_terminated_length": 62.625, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.11215901840478182, "epoch": 0.1006948628348797, "frac_reward_zero_std": 0.0, "grad_norm": 9.360161781311035, "learning_rate": 2.4060606060606062e-06, "loss": 0.023, "num_tokens": 5676923.0, "reward": 0.9846850633621216, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9846850633621216, "reward_meter_std": 0.005829821340739727, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0058298311196267605, "reward_total_composite_mean": 0.9846850633621216, "reward_total_composite_std": 0.005829821340739727, "reward_total_mean": 0.9846850633621216, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9846850633621216, "rewards/meter/std": 0.005829821340739727, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9846850633621216, "rewards/total_composite/std": 0.005829821340739727, "sampling/importance_sampling_ratio/max": 1.7164337635040283, "sampling/importance_sampling_ratio/mean": 0.9932248592376709, "sampling/importance_sampling_ratio/min": 0.00865252036601305, "sampling/sampling_logp_difference/max": 4.749904632568359, "sampling/sampling_logp_difference/mean": 0.048959724605083466, "step": 2507 }, { "clip_ratio/high_max": 0.010315365390852094, "clip_ratio/high_mean": 0.010315365390852094, "clip_ratio/low_mean": 0.010080644860863686, "clip_ratio/low_min": 0.010080644860863686, "clip_ratio/region_mean": 0.02039601025171578, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.125, "completions/mean_terminated_length": 61.125, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.10735112428665161, "epoch": 0.10073502831666466, "frac_reward_zero_std": 0.0, "grad_norm": 2.719905376434326, "learning_rate": 2.403030303030303e-06, "loss": 0.0105, "num_tokens": 5678652.0, "reward": 0.9946359992027283, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9946359992027283, "reward_meter_std": 0.0007906173705123365, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007906315731815994, "reward_total_composite_mean": 0.9946359992027283, "reward_total_composite_std": 0.0007906173705123365, "reward_total_mean": 0.9946359992027283, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9946359992027283, "rewards/meter/std": 0.0007906173705123365, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9946359992027283, "rewards/total_composite/std": 0.0007906173705123365, "sampling/importance_sampling_ratio/max": 1.7869699001312256, "sampling/importance_sampling_ratio/mean": 1.002495288848877, "sampling/importance_sampling_ratio/min": 0.3469068109989166, "sampling/sampling_logp_difference/max": 1.058699131011963, "sampling/sampling_logp_difference/mean": 0.019306229427456856, "step": 2508 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.0018939394503831863, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.01955289370380342, "epoch": 0.10077519379844961, "frac_reward_zero_std": 0.0, "grad_norm": 5.611873626708984, "learning_rate": 2.4000000000000003e-06, "loss": -0.0075, "num_tokens": 5680475.0, "reward": 0.9922016859054565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9922016859054565, "reward_meter_std": 0.016830753535032272, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.016830753535032272, "reward_total_composite_mean": 0.9922016859054565, "reward_total_composite_std": 0.016830753535032272, "reward_total_mean": 0.9922016859054565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9922016859054565, "rewards/meter/std": 0.016830753535032272, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9922016859054565, "rewards/total_composite/std": 0.016830753535032272, "sampling/importance_sampling_ratio/max": 1.0730388164520264, "sampling/importance_sampling_ratio/mean": 0.9988427758216858, "sampling/importance_sampling_ratio/min": 0.21076074242591858, "sampling/sampling_logp_difference/max": 1.5570316314697266, "sampling/sampling_logp_difference/mean": 0.006449920125305653, "step": 2509 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.014974120538681746, "epoch": 0.10081535928023456, "frac_reward_zero_std": 0.0, "grad_norm": 0.017269477248191833, "learning_rate": 2.396969696969697e-06, "loss": -0.0001, "num_tokens": 5682435.0, "reward": 0.9981517195701599, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981517195701599, "reward_meter_std": 1.4962344039304298e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.5052806929816143e-06, "reward_total_composite_mean": 0.9981517195701599, "reward_total_composite_std": 1.4962344039304298e-06, "reward_total_mean": 0.9981517195701599, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981517195701599, "rewards/meter/std": 1.4962344039304298e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981517195701599, "rewards/total_composite/std": 1.4962344039304298e-06, "sampling/importance_sampling_ratio/max": 1.0855460166931152, "sampling/importance_sampling_ratio/mean": 1.0006331205368042, "sampling/importance_sampling_ratio/min": 0.5585451126098633, "sampling/sampling_logp_difference/max": 0.5824198722839355, "sampling/sampling_logp_difference/mean": 0.0025519414339214563, "step": 2510 }, { "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0025510203558951616, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 98.25, "completions/mean_terminated_length": 98.25, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.0416307682171464, "epoch": 0.10085552476201952, "frac_reward_zero_std": 0.0, "grad_norm": 0.2027793526649475, "learning_rate": 2.393939393939394e-06, "loss": 0.0007, "num_tokens": 5684525.0, "reward": 0.9980236291885376, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980236291885376, "reward_meter_std": 2.7453183065517806e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.745408346527256e-05, "reward_total_composite_mean": 0.9980236291885376, "reward_total_composite_std": 2.7453183065517806e-05, "reward_total_mean": 0.9980236291885376, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980236291885376, "rewards/meter/std": 2.7453183065517806e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980236291885376, "rewards/total_composite/std": 2.7453183065517806e-05, "sampling/importance_sampling_ratio/max": 1.3208497762680054, "sampling/importance_sampling_ratio/mean": 1.0008984804153442, "sampling/importance_sampling_ratio/min": 0.2601534426212311, "sampling/sampling_logp_difference/max": 1.3464837074279785, "sampling/sampling_logp_difference/mean": 0.009353289380669594, "step": 2511 }, { "clip_ratio/high_max": 0.020458750892430544, "clip_ratio/high_mean": 0.020458750892430544, "clip_ratio/low_mean": 0.0042372881434857845, "clip_ratio/low_min": 0.0042372881434857845, "clip_ratio/region_mean": 0.02469603903591633, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.11922997329384089, "epoch": 0.10089569024380447, "frac_reward_zero_std": 0.0, "grad_norm": 4.541382312774658, "learning_rate": 2.3909090909090912e-06, "loss": -0.0135, "num_tokens": 5686253.0, "reward": 0.8703480958938599, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948880672454834, "reward_meter_std": 0.0007434968720190227, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.35167405009269714, "reward_total_composite_mean": 0.8703480958938599, "reward_total_composite_std": 0.35167405009269714, "reward_total_mean": 0.8703480958938599, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948880672454834, "rewards/meter/std": 0.0007434968720190227, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8703480958938599, "rewards/total_composite/std": 0.35167405009269714, "sampling/importance_sampling_ratio/max": 1.566518783569336, "sampling/importance_sampling_ratio/mean": 0.9973222613334656, "sampling/importance_sampling_ratio/min": 0.48619309067726135, "sampling/sampling_logp_difference/max": 0.7211494445800781, "sampling/sampling_logp_difference/mean": 0.0175700094550848, "step": 2512 }, { "clip_ratio/high_max": 0.014292625011876225, "clip_ratio/high_mean": 0.014292625011876225, "clip_ratio/low_mean": 0.015607842477038503, "clip_ratio/low_min": 0.015607842477038503, "clip_ratio/region_mean": 0.029900467488914728, "completions/clipped_ratio": 0.0, "completions/max_length": 173.0, "completions/max_terminated_length": 173.0, "completions/mean_length": 166.125, "completions/mean_terminated_length": 166.125, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.37756787799298763, "epoch": 0.10093585572558943, "frac_reward_zero_std": 0.0, "grad_norm": 3.2557570934295654, "learning_rate": 2.387878787878788e-06, "loss": 0.0078, "num_tokens": 5689270.0, "reward": 0.7611844539642334, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9812671542167664, "reward_meter_std": 0.015679193660616875, "reward_repeat_penalty_mean": 0.930555522441864, "reward_repeat_penalty_std": 0.08266931027173996, "reward_std": 0.07191364467144012, "reward_total_composite_mean": 0.7611844539642334, "reward_total_composite_std": 0.07191365212202072, "reward_total_mean": 0.7611844539642334, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9812671542167664, "rewards/meter/std": 0.015679193660616875, "rewards/repeat_penalty/mean": 0.930555522441864, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.7611844539642334, "rewards/total_composite/std": 0.07191365212202072, "sampling/importance_sampling_ratio/max": 1.982582449913025, "sampling/importance_sampling_ratio/mean": 1.0082600116729736, "sampling/importance_sampling_ratio/min": 0.26592347025871277, "sampling/sampling_logp_difference/max": 1.3245468139648438, "sampling/sampling_logp_difference/mean": 0.04680168256163597, "step": 2513 }, { "clip_ratio/high_max": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/region_mean": 0.00390625, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.017428745748475194, "epoch": 0.10097602120737438, "frac_reward_zero_std": 0.0, "grad_norm": 0.11680782586336136, "learning_rate": 2.3848484848484853e-06, "loss": 0.0002, "num_tokens": 5691062.0, "reward": 0.999394416809082, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999394416809082, "reward_meter_std": 3.964109055232257e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.96097129851114e-06, "reward_total_composite_mean": 0.999394416809082, "reward_total_composite_std": 3.964109055232257e-06, "reward_total_mean": 0.999394416809082, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999394416809082, "rewards/meter/std": 3.964109055232257e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999394416809082, "rewards/total_composite/std": 3.964109055232257e-06, "sampling/importance_sampling_ratio/max": 1.4564316272735596, "sampling/importance_sampling_ratio/mean": 1.0025361776351929, "sampling/importance_sampling_ratio/min": 0.6249058246612549, "sampling/sampling_logp_difference/max": 0.47015440464019775, "sampling/sampling_logp_difference/mean": 0.004099253565073013, "step": 2514 }, { "clip_ratio/high_max": 0.009006721433252096, "clip_ratio/high_mean": 0.009006721433252096, "clip_ratio/low_mean": 0.0070080646546557546, "clip_ratio/low_min": 0.0070080646546557546, "clip_ratio/region_mean": 0.01601478608790785, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 124.75, "completions/mean_terminated_length": 124.75, "completions/min_length": 119.0, "completions/min_terminated_length": 119.0, "entropy": 0.14727903716266155, "epoch": 0.10101618668915933, "frac_reward_zero_std": 0.0, "grad_norm": 4.214909076690674, "learning_rate": 2.381818181818182e-06, "loss": 0.0025, "num_tokens": 5693436.0, "reward": 0.9227913618087769, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9937890768051147, "reward_meter_std": 0.000979111879132688, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.07571373879909515, "reward_total_composite_mean": 0.9227913618087769, "reward_total_composite_std": 0.07571373134851456, "reward_total_mean": 0.9227913618087769, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9937890768051147, "rewards/meter/std": 0.000979111879132688, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9227913618087769, "rewards/total_composite/std": 0.07571373134851456, "sampling/importance_sampling_ratio/max": 1.6431286334991455, "sampling/importance_sampling_ratio/mean": 1.0026657581329346, "sampling/importance_sampling_ratio/min": 0.36303985118865967, "sampling/sampling_logp_difference/max": 1.0132427215576172, "sampling/sampling_logp_difference/mean": 0.021321963518857956, "step": 2515 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 106.0, "completions/mean_terminated_length": 106.0, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.0042294226586818695, "epoch": 0.10105635217094429, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.378787878787879e-06, "loss": 0.0, "num_tokens": 5695612.0, "reward": 0.5888487696647644, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6869902610778809, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 0.8571428656578064, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.5888487696647644, "reward_total_composite_std": 0.0, "reward_total_mean": 0.5888487696647644, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6869902610778809, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 0.8571428656578064, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5888487696647644, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0088952779769897, "sampling/importance_sampling_ratio/mean": 1.0005320310592651, "sampling/importance_sampling_ratio/min": 0.9996694326400757, "sampling/sampling_logp_difference/max": 0.00885598175227642, "sampling/sampling_logp_difference/mean": 0.0005320468917489052, "step": 2516 }, { "clip_ratio/high_max": 0.0541982427239418, "clip_ratio/high_mean": 0.0541982427239418, "clip_ratio/low_mean": 0.02324857749044895, "clip_ratio/low_min": 0.02324857749044895, "clip_ratio/region_mean": 0.07744682021439075, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 67.25, "completions/mean_terminated_length": 67.25, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.45786258578300476, "epoch": 0.10109651765272924, "frac_reward_zero_std": 0.0, "grad_norm": 6.815549850463867, "learning_rate": 2.375757575757576e-06, "loss": 0.0787, "num_tokens": 5697454.0, "reward": 0.6705501079559326, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7423630952835083, "reward_meter_std": 0.3543585538864136, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.3163512945175171, "reward_total_composite_mean": 0.6705501079559326, "reward_total_composite_std": 0.3163512647151947, "reward_total_mean": 0.6705501079559326, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7423630952835083, "rewards/meter/std": 0.3543585538864136, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.6705501079559326, "rewards/total_composite/std": 0.3163512647151947, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0203852653503418, "sampling/importance_sampling_ratio/min": 0.15949344635009766, "sampling/sampling_logp_difference/max": 1.8357524871826172, "sampling/sampling_logp_difference/mean": 0.07449356466531754, "step": 2517 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.007087996578775346, "epoch": 0.1011366831345142, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.372727272727273e-06, "loss": 0.0, "num_tokens": 5698926.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0032331943511963, "sampling/importance_sampling_ratio/mean": 0.9997519254684448, "sampling/importance_sampling_ratio/min": 0.9608973860740662, "sampling/sampling_logp_difference/max": 0.0398876816034317, "sampling/sampling_logp_difference/mean": 0.0005153539241291583, "step": 2518 }, { "clip_ratio/high_max": 0.002922117244452238, "clip_ratio/high_mean": 0.002922117244452238, "clip_ratio/low_mean": 0.003992063691839576, "clip_ratio/low_min": 0.003992063691839576, "clip_ratio/region_mean": 0.006914180936291814, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 126.625, "completions/mean_terminated_length": 126.625, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.10657109878957272, "epoch": 0.10117684861629915, "frac_reward_zero_std": 0.0, "grad_norm": 1.6101531982421875, "learning_rate": 2.36969696969697e-06, "loss": -0.0053, "num_tokens": 5701507.0, "reward": 0.8810223340988159, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9869413375854492, "reward_meter_std": 0.010035636834800243, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06303300708532333, "reward_total_composite_mean": 0.8810223340988159, "reward_total_composite_std": 0.06303300708532333, "reward_total_mean": 0.8810223340988159, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9869413375854492, "rewards/meter/std": 0.010035636834800243, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8810223340988159, "rewards/total_composite/std": 0.06303300708532333, "sampling/importance_sampling_ratio/max": 1.6338825225830078, "sampling/importance_sampling_ratio/mean": 1.0051202774047852, "sampling/importance_sampling_ratio/min": 0.458891898393631, "sampling/sampling_logp_difference/max": 0.7789406776428223, "sampling/sampling_logp_difference/mean": 0.012142295949161053, "step": 2519 }, { "clip_ratio/high_max": 0.02055674116127193, "clip_ratio/high_mean": 0.02055674116127193, "clip_ratio/low_mean": 0.013246799120679498, "clip_ratio/low_min": 0.013246799120679498, "clip_ratio/region_mean": 0.03380354028195143, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 36.625, "completions/mean_terminated_length": 36.625, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.44045621529221535, "epoch": 0.1012170140980841, "frac_reward_zero_std": 0.0, "grad_norm": 8.599313735961914, "learning_rate": 2.3666666666666667e-06, "loss": 0.0271, "num_tokens": 5702952.0, "reward": 0.9569580554962158, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9569580554962158, "reward_meter_std": 0.05064847692847252, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.05064847320318222, "reward_total_composite_mean": 0.9569580554962158, "reward_total_composite_std": 0.05064847692847252, "reward_total_mean": 0.9569580554962158, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9569580554962158, "rewards/meter/std": 0.05064847692847252, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9569580554962158, "rewards/total_composite/std": 0.05064847692847252, "sampling/importance_sampling_ratio/max": 1.500638484954834, "sampling/importance_sampling_ratio/mean": 1.0103925466537476, "sampling/importance_sampling_ratio/min": 0.22248442471027374, "sampling/sampling_logp_difference/max": 1.5028982162475586, "sampling/sampling_logp_difference/mean": 0.05431085824966431, "step": 2520 }, { "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.008196720853447914, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.09736423194408417, "epoch": 0.10125717957986906, "frac_reward_zero_std": 0.0, "grad_norm": 4.829682350158691, "learning_rate": 2.363636363636364e-06, "loss": 0.004, "num_tokens": 5704880.0, "reward": 0.994844377040863, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.994844377040863, "reward_meter_std": 0.00046124093933030963, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004612513293977827, "reward_total_composite_mean": 0.994844377040863, "reward_total_composite_std": 0.00046124093933030963, "reward_total_mean": 0.994844377040863, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.994844377040863, "rewards/meter/std": 0.00046124093933030963, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.994844377040863, "rewards/total_composite/std": 0.00046124093933030963, "sampling/importance_sampling_ratio/max": 1.8434381484985352, "sampling/importance_sampling_ratio/mean": 0.9970283508300781, "sampling/importance_sampling_ratio/min": 0.08637453615665436, "sampling/sampling_logp_difference/max": 2.4490623474121094, "sampling/sampling_logp_difference/mean": 0.02323114313185215, "step": 2521 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 55.0, "completions/max_terminated_length": 55.0, "completions/mean_length": 54.125, "completions/mean_terminated_length": 54.125, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0036355706688482314, "epoch": 0.10129734506165401, "frac_reward_zero_std": 0.0, "grad_norm": 0.5622905492782593, "learning_rate": 2.360606060606061e-06, "loss": 0.001, "num_tokens": 5706673.0, "reward": 0.7876288890838623, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7876288890838623, "reward_meter_std": 2.806982047331985e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.8069800464436412e-05, "reward_total_composite_mean": 0.7876288890838623, "reward_total_composite_std": 2.806982047331985e-05, "reward_total_mean": 0.7876288890838623, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7876288890838623, "rewards/meter/std": 2.806982047331985e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7876288890838623, "rewards/total_composite/std": 2.806982047331985e-05, "sampling/importance_sampling_ratio/max": 1.0073615312576294, "sampling/importance_sampling_ratio/mean": 0.9992128610610962, "sampling/importance_sampling_ratio/min": 0.4758167266845703, "sampling/sampling_logp_difference/max": 0.7427225112915039, "sampling/sampling_logp_difference/mean": 0.0021415329538285732, "step": 2522 }, { "clip_ratio/high_max": 0.020694297272711992, "clip_ratio/high_mean": 0.020694297272711992, "clip_ratio/low_mean": 0.009364859783090651, "clip_ratio/low_min": 0.009364859783090651, "clip_ratio/region_mean": 0.030059157055802643, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 497.125, "completions/mean_terminated_length": 495.0000305175781, "completions/min_length": 476.0, "completions/min_terminated_length": 476.0, "entropy": 0.43601541593670845, "epoch": 0.10133751054343897, "frac_reward_zero_std": 0.0, "grad_norm": 1.5801368951797485, "learning_rate": 2.3575757575757577e-06, "loss": 0.099, "num_tokens": 5711794.0, "reward": 0.751766562461853, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8046875, "reward_count_adherence_std": 0.022097086533904076, "reward_meter_mean": 0.9986308813095093, "reward_meter_std": 0.00048286825767718256, "reward_repeat_penalty_mean": 0.9345512986183167, "reward_repeat_penalty_std": 0.06527597457170486, "reward_std": 0.06588190793991089, "reward_total_composite_mean": 0.751766562461853, "reward_total_composite_std": 0.06588190048933029, "reward_total_mean": 0.751766562461853, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8046875, "rewards/count_adherence/std": 0.022097086533904076, "rewards/meter/mean": 0.9986308813095093, "rewards/meter/std": 0.00048286825767718256, "rewards/repeat_penalty/mean": 0.9345512986183167, "rewards/repeat_penalty/std": 0.06527597457170486, "rewards/total_composite/mean": 0.751766562461853, "rewards/total_composite/std": 0.06588190048933029, "sampling/importance_sampling_ratio/max": 1.9566318988800049, "sampling/importance_sampling_ratio/mean": 1.0117627382278442, "sampling/importance_sampling_ratio/min": 0.16628792881965637, "sampling/sampling_logp_difference/max": 1.794034481048584, "sampling/sampling_logp_difference/mean": 0.0517297238111496, "step": 2523 }, { "clip_ratio/high_max": 0.034931490663439035, "clip_ratio/high_mean": 0.034931490663439035, "clip_ratio/low_mean": 0.008616218925453722, "clip_ratio/low_min": 0.008616218925453722, "clip_ratio/region_mean": 0.04354770958889276, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 101.375, "completions/mean_terminated_length": 101.375, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.4374755807220936, "epoch": 0.10137767602522392, "frac_reward_zero_std": 0.0, "grad_norm": 5.143811225891113, "learning_rate": 2.3545454545454545e-06, "loss": 0.0218, "num_tokens": 5713893.0, "reward": 0.9959712028503418, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9959712028503418, "reward_meter_std": 0.005785571411252022, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005785576067864895, "reward_total_composite_mean": 0.9959712028503418, "reward_total_composite_std": 0.005785571411252022, "reward_total_mean": 0.9959712028503418, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9959712028503418, "rewards/meter/std": 0.005785571411252022, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9959712028503418, "rewards/total_composite/std": 0.005785571411252022, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0093885660171509, "sampling/importance_sampling_ratio/min": 0.23741686344146729, "sampling/sampling_logp_difference/max": 1.536201000213623, "sampling/sampling_logp_difference/mean": 0.05520302802324295, "step": 2524 }, { "clip_ratio/high_max": 0.03135894448496401, "clip_ratio/high_mean": 0.03135894448496401, "clip_ratio/low_mean": 0.010438508819788694, "clip_ratio/low_min": 0.010438508819788694, "clip_ratio/region_mean": 0.04179745330475271, "completions/clipped_ratio": 0.0, "completions/max_length": 235.0, "completions/max_terminated_length": 235.0, "completions/mean_length": 227.625, "completions/mean_terminated_length": 227.625, "completions/min_length": 220.0, "completions/min_terminated_length": 220.0, "entropy": 0.5780696794390678, "epoch": 0.10141784150700887, "frac_reward_zero_std": 0.0, "grad_norm": 2.698570966720581, "learning_rate": 2.3515151515151517e-06, "loss": 0.0033, "num_tokens": 5717394.0, "reward": 0.9980570673942566, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980570673942566, "reward_meter_std": 0.0012863370357081294, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0012863423908129334, "reward_total_composite_mean": 0.9980570673942566, "reward_total_composite_std": 0.0012863370357081294, "reward_total_mean": 0.9980570673942566, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980570673942566, "rewards/meter/std": 0.0012863370357081294, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980570673942566, "rewards/total_composite/std": 0.0012863370357081294, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0095667839050293, "sampling/importance_sampling_ratio/min": 0.32984334230422974, "sampling/sampling_logp_difference/max": 1.1091375350952148, "sampling/sampling_logp_difference/mean": 0.06125364825129509, "step": 2525 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.002448402636218816, "epoch": 0.10145800698879383, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.348484848484849e-06, "loss": 0.0, "num_tokens": 5719002.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0036901235580444, "sampling/importance_sampling_ratio/mean": 1.0003321170806885, "sampling/importance_sampling_ratio/min": 0.9999605417251587, "sampling/sampling_logp_difference/max": 0.003683246672153473, "sampling/sampling_logp_difference/mean": 0.0003320652758702636, "step": 2526 }, { "clip_ratio/high_max": 0.00390625, "clip_ratio/high_mean": 0.00390625, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.0078125, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.02413589833304286, "epoch": 0.10149817247057878, "frac_reward_zero_std": 0.0, "grad_norm": 0.11079243570566177, "learning_rate": 2.345454545454546e-06, "loss": 0.0, "num_tokens": 5720778.0, "reward": 0.9993899464607239, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993899464607239, "reward_meter_std": 4.941231964039616e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.955206804879708e-06, "reward_total_composite_mean": 0.9993899464607239, "reward_total_composite_std": 4.941231964039616e-06, "reward_total_mean": 0.9993899464607239, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993899464607239, "rewards/meter/std": 4.941231964039616e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993899464607239, "rewards/total_composite/std": 4.941231964039616e-06, "sampling/importance_sampling_ratio/max": 1.3528425693511963, "sampling/importance_sampling_ratio/mean": 1.0004364252090454, "sampling/importance_sampling_ratio/min": 0.49084919691085815, "sampling/sampling_logp_difference/max": 0.7116183638572693, "sampling/sampling_logp_difference/mean": 0.00580887496471405, "step": 2527 }, { "clip_ratio/high_max": 0.020238142693415284, "clip_ratio/high_mean": 0.020238142693415284, "clip_ratio/low_mean": 0.00951099069789052, "clip_ratio/low_min": 0.00951099069789052, "clip_ratio/region_mean": 0.029749133391305804, "completions/clipped_ratio": 0.0, "completions/max_length": 491.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 483.625, "completions/mean_terminated_length": 483.625, "completions/min_length": 477.0, "completions/min_terminated_length": 477.0, "entropy": 0.2987423837184906, "epoch": 0.10153833795236374, "frac_reward_zero_std": 0.0, "grad_norm": 1.9397426843643188, "learning_rate": 2.3424242424242427e-06, "loss": 0.0113, "num_tokens": 5726703.0, "reward": 0.7824571132659912, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9855457544326782, "reward_meter_std": 0.01022270042449236, "reward_repeat_penalty_mean": 0.9074074029922485, "reward_repeat_penalty_std": 0.059391386806964874, "reward_std": 0.051290810108184814, "reward_total_composite_mean": 0.7824571132659912, "reward_total_composite_std": 0.05129082873463631, "reward_total_mean": 0.7824571132659912, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9855457544326782, "rewards/meter/std": 0.01022270042449236, "rewards/repeat_penalty/mean": 0.9074074029922485, "rewards/repeat_penalty/std": 0.059391386806964874, "rewards/total_composite/mean": 0.7824571132659912, "rewards/total_composite/std": 0.05129082873463631, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0071146488189697, "sampling/importance_sampling_ratio/min": 0.25039762258529663, "sampling/sampling_logp_difference/max": 1.3847050666809082, "sampling/sampling_logp_difference/mean": 0.03337203338742256, "step": 2528 }, { "clip_ratio/high_max": 0.032445638440549374, "clip_ratio/high_mean": 0.032445638440549374, "clip_ratio/low_mean": 0.015321710845455527, "clip_ratio/low_min": 0.015321710845455527, "clip_ratio/region_mean": 0.0477673492860049, "completions/clipped_ratio": 0.0, "completions/max_length": 114.0, "completions/max_terminated_length": 114.0, "completions/mean_length": 107.25, "completions/mean_terminated_length": 107.25, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.5495112352073193, "epoch": 0.10157850343414869, "frac_reward_zero_std": 0.0, "grad_norm": 3.176457405090332, "learning_rate": 2.3393939393939395e-06, "loss": 0.0012, "num_tokens": 5728897.0, "reward": 0.9512283802032471, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9512283802032471, "reward_meter_std": 0.047763749957084656, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04776376485824585, "reward_total_composite_mean": 0.9512283802032471, "reward_total_composite_std": 0.047763749957084656, "reward_total_mean": 0.9512283802032471, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9512283802032471, "rewards/meter/std": 0.047763749957084656, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9512283802032471, "rewards/total_composite/std": 0.047763749957084656, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0175414085388184, "sampling/importance_sampling_ratio/min": 0.2548193633556366, "sampling/sampling_logp_difference/max": 1.3672003746032715, "sampling/sampling_logp_difference/mean": 0.05268417298793793, "step": 2529 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.002742367796599865, "epoch": 0.10161866891593364, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.3363636363636367e-06, "loss": 0.0, "num_tokens": 5730521.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0045462846755981, "sampling/importance_sampling_ratio/mean": 1.0003368854522705, "sampling/importance_sampling_ratio/min": 0.9995154738426208, "sampling/sampling_logp_difference/max": 0.004536015447229147, "sampling/sampling_logp_difference/mean": 0.00033885284210555255, "step": 2530 }, { "clip_ratio/high_max": 0.01107846642844379, "clip_ratio/high_mean": 0.01107846642844379, "clip_ratio/low_mean": 0.01588775822892785, "clip_ratio/low_min": 0.01588775822892785, "clip_ratio/region_mean": 0.02696622465737164, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 79.25, "completions/mean_terminated_length": 79.25, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.3967149890959263, "epoch": 0.1016588343977186, "frac_reward_zero_std": 0.0, "grad_norm": 3.0979435443878174, "learning_rate": 2.3333333333333336e-06, "loss": -0.0018, "num_tokens": 5732411.0, "reward": 0.9984360933303833, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984360933303833, "reward_meter_std": 0.00047055029426701367, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00047055797767825425, "reward_total_composite_mean": 0.9984360933303833, "reward_total_composite_std": 0.00047055029426701367, "reward_total_mean": 0.9984360933303833, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984360933303833, "rewards/meter/std": 0.00047055029426701367, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984360933303833, "rewards/total_composite/std": 0.00047055029426701367, "sampling/importance_sampling_ratio/max": 1.7651348114013672, "sampling/importance_sampling_ratio/mean": 1.0040619373321533, "sampling/importance_sampling_ratio/min": 0.32259345054626465, "sampling/sampling_logp_difference/max": 1.1313624382019043, "sampling/sampling_logp_difference/mean": 0.04553651064634323, "step": 2531 }, { "clip_ratio/high_max": 0.026046376209706068, "clip_ratio/high_mean": 0.026046376209706068, "clip_ratio/low_mean": 0.032748116645962, "clip_ratio/low_min": 0.032748116645962, "clip_ratio/region_mean": 0.05879449285566807, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 74.625, "completions/mean_terminated_length": 74.625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.5188739486038685, "epoch": 0.10169899987950355, "frac_reward_zero_std": 0.0, "grad_norm": 9.73486614227295, "learning_rate": 2.3303030303030304e-06, "loss": 0.0168, "num_tokens": 5734224.0, "reward": 0.3960377871990204, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.3960377871990204, "reward_meter_std": 0.33334165811538696, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.33334165811538696, "reward_total_composite_mean": 0.3960377871990204, "reward_total_composite_std": 0.33334165811538696, "reward_total_mean": 0.3960377871990204, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.3960377871990204, "rewards/meter/std": 0.33334165811538696, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.3960377871990204, "rewards/total_composite/std": 0.33334165811538696, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.011519432067871, "sampling/importance_sampling_ratio/min": 0.20219644904136658, "sampling/sampling_logp_difference/max": 1.598515510559082, "sampling/sampling_logp_difference/mean": 0.0764419287443161, "step": 2532 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.002526927593862638, "epoch": 0.1017391653612885, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.3272727272727277e-06, "loss": 0.0, "num_tokens": 5735968.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.003976583480835, "sampling/importance_sampling_ratio/mean": 1.0003273487091064, "sampling/importance_sampling_ratio/min": 0.999470055103302, "sampling/sampling_logp_difference/max": 0.003968724515289068, "sampling/sampling_logp_difference/mean": 0.00032944101258181036, "step": 2533 }, { "clip_ratio/high_max": 0.017688892083242536, "clip_ratio/high_mean": 0.017688892083242536, "clip_ratio/low_mean": 0.0013440860202535987, "clip_ratio/low_min": 0.0013440860202535987, "clip_ratio/region_mean": 0.019032978103496134, "completions/clipped_ratio": 0.0, "completions/max_length": 95.0, "completions/max_terminated_length": 95.0, "completions/mean_length": 91.5, "completions/mean_terminated_length": 91.5, "completions/min_length": 86.0, "completions/min_terminated_length": 86.0, "entropy": 0.21182428859174252, "epoch": 0.10177933084307346, "frac_reward_zero_std": 0.0, "grad_norm": 2.3901071548461914, "learning_rate": 2.3242424242424245e-06, "loss": 0.0084, "num_tokens": 5738092.0, "reward": 0.9669877886772156, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9918440580368042, "reward_meter_std": 0.003950192127376795, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06943684071302414, "reward_total_composite_mean": 0.9669877886772156, "reward_total_composite_std": 0.06943685561418533, "reward_total_mean": 0.9669877886772156, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9918440580368042, "rewards/meter/std": 0.003950192127376795, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9669877886772156, "rewards/total_composite/std": 0.06943685561418533, "sampling/importance_sampling_ratio/max": 1.6521724462509155, "sampling/importance_sampling_ratio/mean": 1.0052556991577148, "sampling/importance_sampling_ratio/min": 0.4426901638507843, "sampling/sampling_logp_difference/max": 0.814885139465332, "sampling/sampling_logp_difference/mean": 0.030601778998970985, "step": 2534 }, { "clip_ratio/high_max": 0.023648475529626012, "clip_ratio/high_mean": 0.023648475529626012, "clip_ratio/low_mean": 0.006211606087163091, "clip_ratio/low_min": 0.006211606087163091, "clip_ratio/region_mean": 0.029860081616789103, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 511.0, "completions/mean_length": 495.25, "completions/mean_terminated_length": 492.857177734375, "completions/min_length": 479.0, "completions/min_terminated_length": 479.0, "entropy": 0.4710547477006912, "epoch": 0.10181949632485841, "frac_reward_zero_std": 0.0, "grad_norm": 1.6462563276290894, "learning_rate": 2.3212121212121213e-06, "loss": 0.06, "num_tokens": 5743822.0, "reward": 0.7694027423858643, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8046875, "reward_count_adherence_std": 0.022097086533904076, "reward_meter_mean": 0.9958187341690063, "reward_meter_std": 0.006666218861937523, "reward_repeat_penalty_mean": 0.9601762294769287, "reward_repeat_penalty_std": 0.0010186029830947518, "reward_std": 0.021448642015457153, "reward_total_composite_mean": 0.7694027423858643, "reward_total_composite_std": 0.0214486513286829, "reward_total_mean": 0.7694027423858643, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8046875, "rewards/count_adherence/std": 0.022097086533904076, "rewards/meter/mean": 0.9958187341690063, "rewards/meter/std": 0.006666218861937523, "rewards/repeat_penalty/mean": 0.9601762294769287, "rewards/repeat_penalty/std": 0.0010186029830947518, "rewards/total_composite/mean": 0.7694027423858643, "rewards/total_composite/std": 0.0214486513286829, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.014970302581787, "sampling/importance_sampling_ratio/min": 0.11722506582736969, "sampling/sampling_logp_difference/max": 2.1436595916748047, "sampling/sampling_logp_difference/mean": 0.05616435408592224, "step": 2535 }, { "clip_ratio/high_max": 0.052931731566786766, "clip_ratio/high_mean": 0.052931731566786766, "clip_ratio/low_mean": 0.013427280820906162, "clip_ratio/low_min": 0.013427280820906162, "clip_ratio/region_mean": 0.06635901238769293, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 334.75, "completions/mean_terminated_length": 334.75, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "entropy": 0.806172601878643, "epoch": 0.10185966180664337, "frac_reward_zero_std": 0.0, "grad_norm": 3.1345582008361816, "learning_rate": 2.318181818181818e-06, "loss": 0.0168, "num_tokens": 5748428.0, "reward": 0.8046990633010864, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8636363744735718, "reward_count_adherence_std": 0.04859296977519989, "reward_meter_mean": 0.9545778036117554, "reward_meter_std": 0.0988415852189064, "reward_repeat_penalty_mean": 0.9794891476631165, "reward_repeat_penalty_std": 0.028372056782245636, "reward_std": 0.07172780483961105, "reward_total_composite_mean": 0.8046990633010864, "reward_total_composite_std": 0.07172781974077225, "reward_total_mean": 0.8046990633010864, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8636363744735718, "rewards/count_adherence/std": 0.04859296977519989, "rewards/meter/mean": 0.9545778036117554, "rewards/meter/std": 0.0988415852189064, "rewards/repeat_penalty/mean": 0.9794891476631165, "rewards/repeat_penalty/std": 0.028372056782245636, "rewards/total_composite/mean": 0.8046990633010864, "rewards/total_composite/std": 0.07172781974077225, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0162872076034546, "sampling/importance_sampling_ratio/min": 0.00941331684589386, "sampling/sampling_logp_difference/max": 4.665629863739014, "sampling/sampling_logp_difference/mean": 0.08371323347091675, "step": 2536 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0026191577198915184, "epoch": 0.10189982728842832, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.3151515151515154e-06, "loss": 0.0, "num_tokens": 5750148.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0035895109176636, "sampling/importance_sampling_ratio/mean": 1.0003106594085693, "sampling/importance_sampling_ratio/min": 0.999996542930603, "sampling/sampling_logp_difference/max": 0.0035831150598824024, "sampling/sampling_logp_difference/mean": 0.00031052454141899943, "step": 2537 }, { "clip_ratio/high_max": 0.027109916205517948, "clip_ratio/high_mean": 0.027109916205517948, "clip_ratio/low_mean": 0.02077368483878672, "clip_ratio/low_min": 0.02077368483878672, "clip_ratio/region_mean": 0.04788360104430467, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 70.5, "completions/mean_terminated_length": 70.5, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.577742449939251, "epoch": 0.10193999277021328, "frac_reward_zero_std": 0.0, "grad_norm": 8.2296781539917, "learning_rate": 2.3121212121212123e-06, "loss": 0.0261, "num_tokens": 5751904.0, "reward": 0.9975142478942871, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975142478942871, "reward_meter_std": 0.003199664643034339, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0031996748875826597, "reward_total_composite_mean": 0.9975142478942871, "reward_total_composite_std": 0.003199664643034339, "reward_total_mean": 0.9975142478942871, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975142478942871, "rewards/meter/std": 0.003199664643034339, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975142478942871, "rewards/total_composite/std": 0.003199664643034339, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0132718086242676, "sampling/importance_sampling_ratio/min": 0.3238064646720886, "sampling/sampling_logp_difference/max": 1.1276092529296875, "sampling/sampling_logp_difference/mean": 0.06007351353764534, "step": 2538 }, { "clip_ratio/high_max": 0.017147069913335145, "clip_ratio/high_mean": 0.017147069913335145, "clip_ratio/low_mean": 0.005116959102451801, "clip_ratio/low_min": 0.005116959102451801, "clip_ratio/region_mean": 0.022264029015786946, "completions/clipped_ratio": 0.0, "completions/max_length": 76.0, "completions/max_terminated_length": 76.0, "completions/mean_length": 72.5, "completions/mean_terminated_length": 72.5, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.16330011747777462, "epoch": 0.10198015825199823, "frac_reward_zero_std": 0.0, "grad_norm": 3.5843658447265625, "learning_rate": 2.309090909090909e-06, "loss": -0.0026, "num_tokens": 5753708.0, "reward": 0.9982837438583374, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982837438583374, "reward_meter_std": 0.0006137907621450722, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006137940799817443, "reward_total_composite_mean": 0.9982837438583374, "reward_total_composite_std": 0.0006137907621450722, "reward_total_mean": 0.9982837438583374, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982837438583374, "rewards/meter/std": 0.0006137907621450722, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982837438583374, "rewards/total_composite/std": 0.0006137907621450722, "sampling/importance_sampling_ratio/max": 1.6161738634109497, "sampling/importance_sampling_ratio/mean": 1.0038199424743652, "sampling/importance_sampling_ratio/min": 0.33979055285453796, "sampling/sampling_logp_difference/max": 1.0794258117675781, "sampling/sampling_logp_difference/mean": 0.023249905556440353, "step": 2539 }, { "clip_ratio/high_max": 0.017286017769947648, "clip_ratio/high_mean": 0.017286017769947648, "clip_ratio/low_mean": 0.024065676843747497, "clip_ratio/low_min": 0.024065676843747497, "clip_ratio/region_mean": 0.041351694613695145, "completions/clipped_ratio": 0.0, "completions/max_length": 203.0, "completions/max_terminated_length": 203.0, "completions/mean_length": 193.125, "completions/mean_terminated_length": 193.125, "completions/min_length": 184.0, "completions/min_terminated_length": 184.0, "entropy": 0.5925077982246876, "epoch": 0.10202032373378318, "frac_reward_zero_std": 0.0, "grad_norm": 2.297759771347046, "learning_rate": 2.306060606060606e-06, "loss": -0.0035, "num_tokens": 5756765.0, "reward": 0.9978808760643005, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978808760643005, "reward_meter_std": 0.001249481923878193, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0012494820402935147, "reward_total_composite_mean": 0.9978808760643005, "reward_total_composite_std": 0.001249481923878193, "reward_total_mean": 0.9978808760643005, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978808760643005, "rewards/meter/std": 0.001249481923878193, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978808760643005, "rewards/total_composite/std": 0.001249481923878193, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0168569087982178, "sampling/importance_sampling_ratio/min": 0.2672429084777832, "sampling/sampling_logp_difference/max": 1.3195972442626953, "sampling/sampling_logp_difference/mean": 0.05693472549319267, "step": 2540 }, { "clip_ratio/high_max": 0.025725040584802628, "clip_ratio/high_mean": 0.025725040584802628, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/region_mean": 0.02808353118598461, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 107.0, "completions/mean_terminated_length": 107.0, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.19336522929370403, "epoch": 0.10206048921556814, "frac_reward_zero_std": 0.0, "grad_norm": 5.599558353424072, "learning_rate": 2.303030303030303e-06, "loss": 0.0017, "num_tokens": 5758973.0, "reward": 0.9733964800834656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983425140380859, "reward_meter_std": 0.0005217316211201251, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07076217234134674, "reward_total_composite_mean": 0.9733964800834656, "reward_total_composite_std": 0.07076215744018555, "reward_total_mean": 0.9733964800834656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983425140380859, "rewards/meter/std": 0.0005217316211201251, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9733964800834656, "rewards/total_composite/std": 0.07076215744018555, "sampling/importance_sampling_ratio/max": 1.613302230834961, "sampling/importance_sampling_ratio/mean": 1.00425386428833, "sampling/importance_sampling_ratio/min": 0.3031857907772064, "sampling/sampling_logp_difference/max": 1.1934094429016113, "sampling/sampling_logp_difference/mean": 0.02940228208899498, "step": 2541 }, { "clip_ratio/high_max": 0.0078125, "clip_ratio/high_mean": 0.0078125, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/region_mean": 0.009765625, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.02692276705056429, "epoch": 0.10210065469735309, "frac_reward_zero_std": 0.0, "grad_norm": 0.0943424329161644, "learning_rate": 2.3000000000000004e-06, "loss": 0.0001, "num_tokens": 5760901.0, "reward": 0.9993921518325806, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993921518325806, "reward_meter_std": 5.212498763285112e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.226743269304279e-06, "reward_total_composite_mean": 0.9993921518325806, "reward_total_composite_std": 5.212498763285112e-06, "reward_total_mean": 0.9993921518325806, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993921518325806, "rewards/meter/std": 5.212498763285112e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993921518325806, "rewards/total_composite/std": 5.212498763285112e-06, "sampling/importance_sampling_ratio/max": 1.5909435749053955, "sampling/importance_sampling_ratio/mean": 1.0001180171966553, "sampling/importance_sampling_ratio/min": 0.777456521987915, "sampling/sampling_logp_difference/max": 0.464327335357666, "sampling/sampling_logp_difference/mean": 0.004389709793031216, "step": 2542 }, { "clip_ratio/high_max": 0.03669871832244098, "clip_ratio/high_mean": 0.03669871832244098, "clip_ratio/low_mean": 0.0061848959885537624, "clip_ratio/low_min": 0.0061848959885537624, "clip_ratio/region_mean": 0.042883614310994744, "completions/clipped_ratio": 0.0, "completions/max_length": 384.0, "completions/max_terminated_length": 384.0, "completions/mean_length": 359.5, "completions/mean_terminated_length": 359.5, "completions/min_length": 338.0, "completions/min_terminated_length": 338.0, "entropy": 0.6531303524971008, "epoch": 0.10214082017913804, "frac_reward_zero_std": 0.0, "grad_norm": 2.6460602283477783, "learning_rate": 2.2969696969696973e-06, "loss": 0.0244, "num_tokens": 5765713.0, "reward": 0.7973617315292358, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.05175493285059929, "reward_meter_mean": 0.9972070455551147, "reward_meter_std": 0.0018466432811692357, "reward_repeat_penalty_mean": 0.9864766001701355, "reward_repeat_penalty_std": 0.025052649900317192, "reward_std": 0.32495421171188354, "reward_total_composite_mean": 0.7973617315292358, "reward_total_composite_std": 0.32495424151420593, "reward_total_mean": 0.7973617315292358, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.05175493285059929, "rewards/meter/mean": 0.9972070455551147, "rewards/meter/std": 0.0018466432811692357, "rewards/repeat_penalty/mean": 0.9864766001701355, "rewards/repeat_penalty/std": 0.025052649900317192, "rewards/total_composite/mean": 0.7973617315292358, "rewards/total_composite/std": 0.32495424151420593, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0157252550125122, "sampling/importance_sampling_ratio/min": 0.22023677825927734, "sampling/sampling_logp_difference/max": 1.513051986694336, "sampling/sampling_logp_difference/mean": 0.07025956362485886, "step": 2543 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0035496088094078004, "epoch": 0.102180985660923, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.293939393939394e-06, "loss": 0.0, "num_tokens": 5767393.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0081822872161865, "sampling/importance_sampling_ratio/mean": 1.000386118888855, "sampling/importance_sampling_ratio/min": 0.9980019927024841, "sampling/sampling_logp_difference/max": 0.008148963563144207, "sampling/sampling_logp_difference/mean": 0.0003962568298447877, "step": 2544 }, { "clip_ratio/high_max": 0.004065309185534716, "clip_ratio/high_mean": 0.004065309185534716, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.006114489398896694, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.125, "completions/mean_terminated_length": 61.125, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.09053080063313246, "epoch": 0.10222115114270795, "frac_reward_zero_std": 0.0, "grad_norm": 3.4628183841705322, "learning_rate": 2.2909090909090913e-06, "loss": -0.0032, "num_tokens": 5769250.0, "reward": 0.9939138889312744, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939138889312744, "reward_meter_std": 0.0031664494890719652, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0031664466951042414, "reward_total_composite_mean": 0.9939138889312744, "reward_total_composite_std": 0.0031664494890719652, "reward_total_mean": 0.9939138889312744, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939138889312744, "rewards/meter/std": 0.0031664494890719652, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9939138889312744, "rewards/total_composite/std": 0.0031664494890719652, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004980444908142, "sampling/importance_sampling_ratio/min": 0.14398977160453796, "sampling/sampling_logp_difference/max": 1.9380130767822266, "sampling/sampling_logp_difference/mean": 0.020268075168132782, "step": 2545 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.0018656715983524919, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.025023984955623746, "epoch": 0.1022613166244929, "frac_reward_zero_std": 0.0, "grad_norm": 0.5305520296096802, "learning_rate": 2.287878787878788e-06, "loss": -0.0005, "num_tokens": 5771130.0, "reward": 0.9981376528739929, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981376528739929, "reward_meter_std": 2.376116935920436e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.3771623091306537e-05, "reward_total_composite_mean": 0.9981376528739929, "reward_total_composite_std": 2.376116935920436e-05, "reward_total_mean": 0.9981376528739929, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981376528739929, "rewards/meter/std": 2.376116935920436e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981376528739929, "rewards/total_composite/std": 2.376116935920436e-05, "sampling/importance_sampling_ratio/max": 1.1664998531341553, "sampling/importance_sampling_ratio/mean": 0.9998449087142944, "sampling/importance_sampling_ratio/min": 0.6188273429870605, "sampling/sampling_logp_difference/max": 0.47992897033691406, "sampling/sampling_logp_difference/mean": 0.004880668129771948, "step": 2546 }, { "clip_ratio/high_max": 0.05134916538372636, "clip_ratio/high_mean": 0.05134916538372636, "clip_ratio/low_mean": 0.008633634075522423, "clip_ratio/low_min": 0.008633634075522423, "clip_ratio/region_mean": 0.05998279945924878, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 322.375, "completions/mean_terminated_length": 322.375, "completions/min_length": 287.0, "completions/min_terminated_length": 287.0, "entropy": 0.9364352449774742, "epoch": 0.10230148210627786, "frac_reward_zero_std": 0.0, "grad_norm": 3.3049399852752686, "learning_rate": 2.284848484848485e-06, "loss": 0.0142, "num_tokens": 5775349.0, "reward": 0.6906000375747681, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8020833134651184, "reward_count_adherence_std": 0.0431290864944458, "reward_meter_mean": 0.9897097945213318, "reward_meter_std": 0.02135160192847252, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2835569977760315, "reward_total_composite_mean": 0.6906000375747681, "reward_total_composite_std": 0.2835569977760315, "reward_total_mean": 0.6906000375747681, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8020833134651184, "rewards/count_adherence/std": 0.0431290864944458, "rewards/meter/mean": 0.9897097945213318, "rewards/meter/std": 0.02135160192847252, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6906000375747681, "rewards/total_composite/std": 0.2835569977760315, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0231871604919434, "sampling/importance_sampling_ratio/min": 0.15247458219528198, "sampling/sampling_logp_difference/max": 1.8807573318481445, "sampling/sampling_logp_difference/mean": 0.09224426746368408, "step": 2547 }, { "clip_ratio/high_max": 0.06528556300327182, "clip_ratio/high_mean": 0.06528556300327182, "clip_ratio/low_mean": 0.00866977241821587, "clip_ratio/low_min": 0.00866977241821587, "clip_ratio/region_mean": 0.07395533542148769, "completions/clipped_ratio": 0.0, "completions/max_length": 46.0, "completions/max_terminated_length": 46.0, "completions/mean_length": 42.5, "completions/mean_terminated_length": 42.5, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.5177558250725269, "epoch": 0.10234164758806281, "frac_reward_zero_std": 0.0, "grad_norm": 13.814697265625, "learning_rate": 2.281818181818182e-06, "loss": 0.0725, "num_tokens": 5776993.0, "reward": 0.9308160543441772, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9308160543441772, "reward_meter_std": 0.01748719997704029, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0174871776252985, "reward_total_composite_mean": 0.9308160543441772, "reward_total_composite_std": 0.01748719997704029, "reward_total_mean": 0.9308160543441772, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9308160543441772, "rewards/meter/std": 0.01748719997704029, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9308160543441772, "rewards/total_composite/std": 0.01748719997704029, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9965495467185974, "sampling/importance_sampling_ratio/min": 0.08522769808769226, "sampling/sampling_logp_difference/max": 2.4624288082122803, "sampling/sampling_logp_difference/mean": 0.0993395447731018, "step": 2548 }, { "clip_ratio/high_max": 0.023415555711835623, "clip_ratio/high_mean": 0.023415555711835623, "clip_ratio/low_mean": 0.014413215219974518, "clip_ratio/low_min": 0.014413215219974518, "clip_ratio/region_mean": 0.03782877093181014, "completions/clipped_ratio": 0.0, "completions/max_length": 196.0, "completions/max_terminated_length": 196.0, "completions/mean_length": 191.875, "completions/mean_terminated_length": 191.875, "completions/min_length": 187.0, "completions/min_terminated_length": 187.0, "entropy": 0.4880506880581379, "epoch": 0.10238181306984777, "frac_reward_zero_std": 0.0, "grad_norm": 1.8126007318496704, "learning_rate": 2.278787878787879e-06, "loss": 0.0034, "num_tokens": 5780008.0, "reward": 0.9984524250030518, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984524250030518, "reward_meter_std": 0.0005822303937748075, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005822345847263932, "reward_total_composite_mean": 0.9984524250030518, "reward_total_composite_std": 0.0005822303937748075, "reward_total_mean": 0.9984524250030518, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984524250030518, "rewards/meter/std": 0.0005822303937748075, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984524250030518, "rewards/total_composite/std": 0.0005822303937748075, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0145859718322754, "sampling/importance_sampling_ratio/min": 0.3044286072254181, "sampling/sampling_logp_difference/max": 1.1893186569213867, "sampling/sampling_logp_difference/mean": 0.055033281445503235, "step": 2549 }, { "clip_ratio/high_max": 0.020960328169167042, "clip_ratio/high_mean": 0.020960328169167042, "clip_ratio/low_mean": 0.017614107578992844, "clip_ratio/low_min": 0.017614107578992844, "clip_ratio/region_mean": 0.038574435748159885, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 68.25, "completions/mean_terminated_length": 68.25, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.5948160253465176, "epoch": 0.10242197855163272, "frac_reward_zero_std": 0.0, "grad_norm": 4.8445634841918945, "learning_rate": 2.275757575757576e-06, "loss": 0.0457, "num_tokens": 5781794.0, "reward": 0.980836033821106, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.980836033821106, "reward_meter_std": 0.014489145018160343, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014489145949482918, "reward_total_composite_mean": 0.980836033821106, "reward_total_composite_std": 0.014489145018160343, "reward_total_mean": 0.980836033821106, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.980836033821106, "rewards/meter/std": 0.014489145018160343, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.980836033821106, "rewards/total_composite/std": 0.014489145018160343, "sampling/importance_sampling_ratio/max": 1.8987267017364502, "sampling/importance_sampling_ratio/mean": 1.006621241569519, "sampling/importance_sampling_ratio/min": 0.20719482004642487, "sampling/sampling_logp_difference/max": 1.5740957260131836, "sampling/sampling_logp_difference/mean": 0.05946962162852287, "step": 2550 }, { "epoch": 0.10242197855163272, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/max_length": 409.0769230769231, "eval_completions/max_terminated_length": 399.2307692307692, "eval_completions/mean_length": 210.5, "eval_completions/mean_terminated_length": 207.71153963529147, "eval_completions/min_length": 62.92307692307692, "eval_completions/min_terminated_length": 62.92307692307692, "eval_entropy": 0.42294850601599765, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 5781794.0, "eval_reward": 0.6745542379525992, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.08158924258672275, "eval_reward_count_adherence_mean": 0.9420714332507207, "eval_reward_count_adherence_std": 0.08270380875239006, "eval_reward_meter_mean": 0.768360605606666, "eval_reward_meter_std": 0.34562337971650636, "eval_reward_repeat_penalty_mean": 0.9319783999369695, "eval_reward_repeat_penalty_std": 0.09335804444092971, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6745542379525992, "eval_reward_total_composite_std": 0.3481578013071647, "eval_reward_total_mean": 0.6745542379525992, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.08158924258672275, "eval_rewards/count_adherence/mean": 0.9420714332507207, "eval_rewards/count_adherence/std": 0.08270380875239006, "eval_rewards/meter/mean": 0.768360605606666, "eval_rewards/meter/std": 0.34562337971650636, "eval_rewards/repeat_penalty/mean": 0.9319783999369695, "eval_rewards/repeat_penalty/std": 0.09335804444092971, "eval_rewards/total_composite/mean": 0.6745542379525992, "eval_rewards/total_composite/std": 0.3481578013071647, "eval_runtime": 76.2084, "eval_samples_per_second": 1.365, "eval_sampling/importance_sampling_ratio/max": 1.5434642296570997, "eval_sampling/importance_sampling_ratio/mean": 1.0112521740106435, "eval_sampling/importance_sampling_ratio/min": 0.29530371954807866, "eval_sampling/sampling_logp_difference/max": 1.2396357609675481, "eval_sampling/sampling_logp_difference/mean": 0.03692194346625071, "eval_steps_per_second": 0.171, "step": 2550 }, { "clip_ratio/high_max": 0.013518144143745303, "clip_ratio/high_mean": 0.013518144143745303, "clip_ratio/low_mean": 0.005239521153271198, "clip_ratio/low_min": 0.005239521153271198, "clip_ratio/region_mean": 0.0187576652970165, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 166.625, "completions/mean_terminated_length": 166.625, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.1492440328001976, "epoch": 0.10246214403341768, "frac_reward_zero_std": 0.0, "grad_norm": 1.7443753480911255, "learning_rate": 2.2727272727272728e-06, "loss": 0.004, "num_tokens": 5784519.0, "reward": 0.998813271522522, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998813271522522, "reward_meter_std": 0.00025403356994502246, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00025402315077371895, "reward_total_composite_mean": 0.998813271522522, "reward_total_composite_std": 0.00025403356994502246, "reward_total_mean": 0.998813271522522, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998813271522522, "rewards/meter/std": 0.00025403356994502246, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998813271522522, "rewards/total_composite/std": 0.00025403356994502246, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0034915208816528, "sampling/importance_sampling_ratio/min": 0.32028141617774963, "sampling/sampling_logp_difference/max": 1.1385552883148193, "sampling/sampling_logp_difference/mean": 0.021267293021082878, "step": 2551 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/region_mean": 0.0012755101779475808, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 98.125, "completions/mean_terminated_length": 98.125, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.05678324867039919, "epoch": 0.10250230951520263, "frac_reward_zero_std": 0.0, "grad_norm": 0.3295544683933258, "learning_rate": 2.2696969696969696e-06, "loss": 0.0005, "num_tokens": 5786536.0, "reward": 0.9980067014694214, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980067014694214, "reward_meter_std": 2.103978840750642e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.104529266944155e-05, "reward_total_composite_mean": 0.9980067014694214, "reward_total_composite_std": 2.103978840750642e-05, "reward_total_mean": 0.9980067014694214, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980067014694214, "rewards/meter/std": 2.103978840750642e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980067014694214, "rewards/total_composite/std": 2.103978840750642e-05, "sampling/importance_sampling_ratio/max": 1.583804726600647, "sampling/importance_sampling_ratio/mean": 1.0038623809814453, "sampling/importance_sampling_ratio/min": 0.20005254447460175, "sampling/sampling_logp_difference/max": 1.609175205230713, "sampling/sampling_logp_difference/mean": 0.008558080531656742, "step": 2552 }, { "clip_ratio/high_max": 0.08001901814714074, "clip_ratio/high_mean": 0.08001901814714074, "clip_ratio/low_mean": 0.028055555652827024, "clip_ratio/low_min": 0.028055555652827024, "clip_ratio/region_mean": 0.10807457379996777, "completions/clipped_ratio": 0.0, "completions/max_length": 51.0, "completions/max_terminated_length": 51.0, "completions/mean_length": 45.25, "completions/mean_terminated_length": 45.25, "completions/min_length": 38.0, "completions/min_terminated_length": 38.0, "entropy": 0.5326411537826061, "epoch": 0.10254247499698758, "frac_reward_zero_std": 0.0, "grad_norm": 12.867728233337402, "learning_rate": 2.266666666666667e-06, "loss": 0.0792, "num_tokens": 5788162.0, "reward": 0.5925682783126831, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.5925682783126831, "reward_meter_std": 0.4251895844936371, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4251895546913147, "reward_total_composite_mean": 0.5925682783126831, "reward_total_composite_std": 0.4251895844936371, "reward_total_mean": 0.5925682783126831, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.5925682783126831, "rewards/meter/std": 0.4251895844936371, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5925682783126831, "rewards/total_composite/std": 0.4251895844936371, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9826568365097046, "sampling/importance_sampling_ratio/min": 0.008494059555232525, "sampling/sampling_logp_difference/max": 4.768388271331787, "sampling/sampling_logp_difference/mean": 0.1222197413444519, "step": 2553 }, { "clip_ratio/high_max": 0.011097930022515357, "clip_ratio/high_mean": 0.011097930022515357, "clip_ratio/low_mean": 0.009753433521836996, "clip_ratio/low_min": 0.009753433521836996, "clip_ratio/region_mean": 0.020851363544352353, "completions/clipped_ratio": 0.0, "completions/max_length": 91.0, "completions/max_terminated_length": 91.0, "completions/mean_length": 89.625, "completions/mean_terminated_length": 89.625, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.2214542292058468, "epoch": 0.10258264047877254, "frac_reward_zero_std": 0.0, "grad_norm": 3.188581705093384, "learning_rate": 2.2636363636363637e-06, "loss": -0.0001, "num_tokens": 5790271.0, "reward": 0.9921834468841553, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9921834468841553, "reward_meter_std": 0.003569816704839468, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0035698034334927797, "reward_total_composite_mean": 0.9921834468841553, "reward_total_composite_std": 0.003569816704839468, "reward_total_mean": 0.9921834468841553, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9921834468841553, "rewards/meter/std": 0.003569816704839468, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9921834468841553, "rewards/total_composite/std": 0.003569816704839468, "sampling/importance_sampling_ratio/max": 1.3844112157821655, "sampling/importance_sampling_ratio/mean": 1.0064231157302856, "sampling/importance_sampling_ratio/min": 0.39650243520736694, "sampling/sampling_logp_difference/max": 0.9250731468200684, "sampling/sampling_logp_difference/mean": 0.032123863697052, "step": 2554 }, { "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/low_mean": 0.005063388962298632, "clip_ratio/low_min": 0.005063388962298632, "clip_ratio/region_mean": 0.007614409318193793, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 98.75, "completions/mean_terminated_length": 98.75, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.042933082208037376, "epoch": 0.10262280596055749, "frac_reward_zero_std": 0.0, "grad_norm": 0.22581298649311066, "learning_rate": 2.260606060606061e-06, "loss": 0.0004, "num_tokens": 5792325.0, "reward": 0.9993069171905518, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993069171905518, "reward_meter_std": 1.8531554815126583e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.8528446162235923e-05, "reward_total_composite_mean": 0.9993069171905518, "reward_total_composite_std": 1.8531554815126583e-05, "reward_total_mean": 0.9993069171905518, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993069171905518, "rewards/meter/std": 1.8531554815126583e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993069171905518, "rewards/total_composite/std": 1.8531554815126583e-05, "sampling/importance_sampling_ratio/max": 1.5615042448043823, "sampling/importance_sampling_ratio/mean": 1.0017414093017578, "sampling/importance_sampling_ratio/min": 0.5430828332901001, "sampling/sampling_logp_difference/max": 0.6104934215545654, "sampling/sampling_logp_difference/mean": 0.008686289191246033, "step": 2555 }, { "clip_ratio/high_max": 0.0364626687951386, "clip_ratio/high_mean": 0.0364626687951386, "clip_ratio/low_mean": 0.008094559656456113, "clip_ratio/low_min": 0.008094559656456113, "clip_ratio/region_mean": 0.04455722845159471, "completions/clipped_ratio": 0.0, "completions/max_length": 209.0, "completions/max_terminated_length": 209.0, "completions/mean_length": 199.125, "completions/mean_terminated_length": 199.125, "completions/min_length": 188.0, "completions/min_terminated_length": 188.0, "entropy": 0.47996126115322113, "epoch": 0.10266297144234245, "frac_reward_zero_std": 0.0, "grad_norm": 2.697803020477295, "learning_rate": 2.2575757575757578e-06, "loss": 0.0125, "num_tokens": 5795614.0, "reward": 0.8613824248313904, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9469772577285767, "reward_meter_std": 0.0619339719414711, "reward_repeat_penalty_mean": 0.9090909361839294, "reward_repeat_penalty_std": 0.08416546136140823, "reward_std": 0.10320668667554855, "reward_total_composite_mean": 0.8613824248313904, "reward_total_composite_std": 0.10320668667554855, "reward_total_mean": 0.8613824248313904, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9469772577285767, "rewards/meter/std": 0.0619339719414711, "rewards/repeat_penalty/mean": 0.9090909361839294, "rewards/repeat_penalty/std": 0.08416546136140823, "rewards/total_composite/mean": 0.8613824248313904, "rewards/total_composite/std": 0.10320668667554855, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.012196660041809, "sampling/importance_sampling_ratio/min": 0.2267685979604721, "sampling/sampling_logp_difference/max": 1.4838252067565918, "sampling/sampling_logp_difference/mean": 0.04738626256585121, "step": 2556 }, { "clip_ratio/high_max": 0.012197049451060593, "clip_ratio/high_mean": 0.012197049451060593, "clip_ratio/low_mean": 0.01822700910270214, "clip_ratio/low_min": 0.01822700910270214, "clip_ratio/region_mean": 0.030424058553762734, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.4355851337313652, "epoch": 0.1027031369241274, "frac_reward_zero_std": 0.0, "grad_norm": 4.737760066986084, "learning_rate": 2.254545454545455e-06, "loss": 0.0071, "num_tokens": 5797542.0, "reward": 0.974303126335144, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.974303126335144, "reward_meter_std": 0.020303523167967796, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.020303526893258095, "reward_total_composite_mean": 0.974303126335144, "reward_total_composite_std": 0.020303523167967796, "reward_total_mean": 0.974303126335144, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.974303126335144, "rewards/meter/std": 0.020303523167967796, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.974303126335144, "rewards/total_composite/std": 0.020303523167967796, "sampling/importance_sampling_ratio/max": 1.927941083908081, "sampling/importance_sampling_ratio/mean": 1.0142627954483032, "sampling/importance_sampling_ratio/min": 0.19931812584400177, "sampling/sampling_logp_difference/max": 1.6128530502319336, "sampling/sampling_logp_difference/mean": 0.04797229915857315, "step": 2557 }, { "clip_ratio/high_max": 0.020191184477880597, "clip_ratio/high_mean": 0.020191184477880597, "clip_ratio/low_mean": 0.022996979532763362, "clip_ratio/low_min": 0.022996979532763362, "clip_ratio/region_mean": 0.04318816401064396, "completions/clipped_ratio": 0.0, "completions/max_length": 139.0, "completions/max_terminated_length": 139.0, "completions/mean_length": 136.25, "completions/mean_terminated_length": 136.25, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.4538475573062897, "epoch": 0.10274330240591235, "frac_reward_zero_std": 0.0, "grad_norm": 3.3955912590026855, "learning_rate": 2.251515151515152e-06, "loss": 0.0093, "num_tokens": 5799904.0, "reward": 0.9227904081344604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.975925624370575, "reward_meter_std": 0.038366202265024185, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07010588049888611, "reward_total_composite_mean": 0.9227904081344604, "reward_total_composite_std": 0.0701058879494667, "reward_total_mean": 0.9227904081344604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.975925624370575, "rewards/meter/std": 0.038366202265024185, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9227904081344604, "rewards/total_composite/std": 0.0701058879494667, "sampling/importance_sampling_ratio/max": 1.85683012008667, "sampling/importance_sampling_ratio/mean": 1.006500005722046, "sampling/importance_sampling_ratio/min": 0.1501615196466446, "sampling/sampling_logp_difference/max": 1.8960437774658203, "sampling/sampling_logp_difference/mean": 0.05260032042860985, "step": 2558 }, { "clip_ratio/high_max": 0.04882641462609172, "clip_ratio/high_mean": 0.04882641462609172, "clip_ratio/low_mean": 0.030968211824074388, "clip_ratio/low_min": 0.030968211824074388, "clip_ratio/region_mean": 0.0797946264501661, "completions/clipped_ratio": 0.0, "completions/max_length": 53.0, "completions/max_terminated_length": 53.0, "completions/mean_length": 44.625, "completions/mean_terminated_length": 44.625, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.5555657297372818, "epoch": 0.10278346788769731, "frac_reward_zero_std": 0.0, "grad_norm": 12.212129592895508, "learning_rate": 2.2484848484848487e-06, "loss": -0.0096, "num_tokens": 5801541.0, "reward": 0.8007308840751648, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8007308840751648, "reward_meter_std": 0.19306603074073792, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.19306600093841553, "reward_total_composite_mean": 0.8007308840751648, "reward_total_composite_std": 0.19306603074073792, "reward_total_mean": 0.8007308840751648, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8007308840751648, "rewards/meter/std": 0.19306603074073792, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8007308840751648, "rewards/total_composite/std": 0.19306603074073792, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9941558837890625, "sampling/importance_sampling_ratio/min": 0.25600266456604004, "sampling/sampling_logp_difference/max": 1.36256742477417, "sampling/sampling_logp_difference/mean": 0.09867885708808899, "step": 2559 }, { "clip_ratio/high_max": 0.014688471332192421, "clip_ratio/high_mean": 0.014688471332192421, "clip_ratio/low_mean": 0.014391222270205617, "clip_ratio/low_min": 0.014391222270205617, "clip_ratio/region_mean": 0.029079693602398038, "completions/clipped_ratio": 0.0, "completions/max_length": 361.0, "completions/max_terminated_length": 361.0, "completions/mean_length": 347.75, "completions/mean_terminated_length": 347.75, "completions/min_length": 337.0, "completions/min_terminated_length": 337.0, "entropy": 0.4555246904492378, "epoch": 0.10282363336948226, "frac_reward_zero_std": 0.0, "grad_norm": 1.910988211631775, "learning_rate": 2.2454545454545455e-06, "loss": -0.0056, "num_tokens": 5806083.0, "reward": 0.7107186317443848, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986293315887451, "reward_meter_std": 0.00039406708674505353, "reward_repeat_penalty_mean": 0.9489378929138184, "reward_repeat_penalty_std": 0.05824849754571915, "reward_std": 0.04346400871872902, "reward_total_composite_mean": 0.7107186317443848, "reward_total_composite_std": 0.043464019894599915, "reward_total_mean": 0.7107186317443848, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986293315887451, "rewards/meter/std": 0.00039406708674505353, "rewards/repeat_penalty/mean": 0.9489378929138184, "rewards/repeat_penalty/std": 0.05824849754571915, "rewards/total_composite/mean": 0.7107186317443848, "rewards/total_composite/std": 0.043464019894599915, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0138837099075317, "sampling/importance_sampling_ratio/min": 4.6412063170464535e-07, "sampling/sampling_logp_difference/max": 14.583121299743652, "sampling/sampling_logp_difference/mean": 0.061371464282274246, "step": 2560 }, { "clip_ratio/high_max": 0.022279977099969983, "clip_ratio/high_mean": 0.022279977099969983, "clip_ratio/low_mean": 0.007634902372956276, "clip_ratio/low_min": 0.007634902372956276, "clip_ratio/region_mean": 0.02991487947292626, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.28918380104005337, "epoch": 0.10286379885126722, "frac_reward_zero_std": 0.0, "grad_norm": 3.3982653617858887, "learning_rate": 2.2424242424242428e-06, "loss": -0.0045, "num_tokens": 5807836.0, "reward": 0.9521244764328003, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9521244764328003, "reward_meter_std": 0.04941366985440254, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.049413666129112244, "reward_total_composite_mean": 0.9521244764328003, "reward_total_composite_std": 0.04941366985440254, "reward_total_mean": 0.9521244764328003, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9521244764328003, "rewards/meter/std": 0.04941366985440254, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9521244764328003, "rewards/total_composite/std": 0.04941366985440254, "sampling/importance_sampling_ratio/max": 1.7526792287826538, "sampling/importance_sampling_ratio/mean": 1.0056248903274536, "sampling/importance_sampling_ratio/min": 0.12441255152225494, "sampling/sampling_logp_difference/max": 2.0841522216796875, "sampling/sampling_logp_difference/mean": 0.03345394507050514, "step": 2561 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.002722353514400311, "epoch": 0.10290396433305217, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.2393939393939396e-06, "loss": 0.0, "num_tokens": 5809388.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0079954862594604, "sampling/importance_sampling_ratio/mean": 1.0001332759857178, "sampling/importance_sampling_ratio/min": 0.9975297451019287, "sampling/sampling_logp_difference/max": 0.007963716052472591, "sampling/sampling_logp_difference/mean": 0.00015674103633500636, "step": 2562 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0025233181950170547, "epoch": 0.10294412981483712, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.2363636363636364e-06, "loss": 0.0, "num_tokens": 5811516.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0044771432876587, "sampling/importance_sampling_ratio/mean": 1.000300645828247, "sampling/importance_sampling_ratio/min": 0.999937891960144, "sampling/sampling_logp_difference/max": 0.00446719815954566, "sampling/sampling_logp_difference/mean": 0.00030092071392573416, "step": 2563 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/region_mean": 0.0033783784601837397, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 37.125, "completions/mean_terminated_length": 37.125, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.059151682537049055, "epoch": 0.10298429529662208, "frac_reward_zero_std": 0.0, "grad_norm": 1.6162538528442383, "learning_rate": 2.2333333333333333e-06, "loss": 0.0027, "num_tokens": 5813045.0, "reward": 0.9993616938591003, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993616938591003, "reward_meter_std": 0.0003300250100437552, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00033003126736730337, "reward_total_composite_mean": 0.9993616938591003, "reward_total_composite_std": 0.0003300250100437552, "reward_total_mean": 0.9993616938591003, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993616938591003, "rewards/meter/std": 0.0003300250100437552, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993616938591003, "rewards/total_composite/std": 0.0003300250100437552, "sampling/importance_sampling_ratio/max": 1.1445908546447754, "sampling/importance_sampling_ratio/mean": 0.9978834986686707, "sampling/importance_sampling_ratio/min": 0.28149527311325073, "sampling/sampling_logp_difference/max": 1.2676396369934082, "sampling/sampling_logp_difference/mean": 0.01185247115790844, "step": 2564 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.004310344811528921, "clip_ratio/low_min": 0.004310344811528921, "clip_ratio/region_mean": 0.004310344811528921, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.007859309727791697, "epoch": 0.10302446077840703, "frac_reward_zero_std": 0.0, "grad_norm": 0.15634091198444366, "learning_rate": 2.2303030303030305e-06, "loss": -0.0009, "num_tokens": 5814621.0, "reward": 0.9957231283187866, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957231283187866, "reward_meter_std": 5.816265183966607e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.816265183966607e-05, "reward_total_composite_mean": 0.9957231283187866, "reward_total_composite_std": 5.816265183966607e-05, "reward_total_mean": 0.9957231283187866, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957231283187866, "rewards/meter/std": 5.816265183966607e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957231283187866, "rewards/total_composite/std": 5.816265183966607e-05, "sampling/importance_sampling_ratio/max": 1.0548250675201416, "sampling/importance_sampling_ratio/mean": 0.99995356798172, "sampling/importance_sampling_ratio/min": 0.8801833987236023, "sampling/sampling_logp_difference/max": 0.1276249885559082, "sampling/sampling_logp_difference/mean": 0.0017511904006823897, "step": 2565 }, { "clip_ratio/high_max": 0.014201456913724542, "clip_ratio/high_mean": 0.014201456913724542, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/region_mean": 0.015824833535589278, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 78.75, "completions/mean_terminated_length": 78.75, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.22580487374216318, "epoch": 0.10306462626019199, "frac_reward_zero_std": 0.0, "grad_norm": 1.9842983484268188, "learning_rate": 2.2272727272727274e-06, "loss": -0.0106, "num_tokens": 5816563.0, "reward": 0.99830561876297, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99830561876297, "reward_meter_std": 0.001100848545320332, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0011008477304130793, "reward_total_composite_mean": 0.99830561876297, "reward_total_composite_std": 0.001100848545320332, "reward_total_mean": 0.99830561876297, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99830561876297, "rewards/meter/std": 0.001100848545320332, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99830561876297, "rewards/total_composite/std": 0.001100848545320332, "sampling/importance_sampling_ratio/max": 1.649073600769043, "sampling/importance_sampling_ratio/mean": 1.0084333419799805, "sampling/importance_sampling_ratio/min": 0.37935489416122437, "sampling/sampling_logp_difference/max": 0.9692831039428711, "sampling/sampling_logp_difference/mean": 0.02984953485429287, "step": 2566 }, { "clip_ratio/high_max": 0.030598473269492388, "clip_ratio/high_mean": 0.030598473269492388, "clip_ratio/low_mean": 0.015345451422035694, "clip_ratio/low_min": 0.015345451422035694, "clip_ratio/region_mean": 0.04594392469152808, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.375, "completions/mean_terminated_length": 70.375, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.41449311561882496, "epoch": 0.10310479174197694, "frac_reward_zero_std": 0.0, "grad_norm": 5.494159698486328, "learning_rate": 2.224242424242424e-06, "loss": 0.03, "num_tokens": 5818390.0, "reward": 0.9767506122589111, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9767506122589111, "reward_meter_std": 0.017817148938775063, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.017817148938775063, "reward_total_composite_mean": 0.9767506122589111, "reward_total_composite_std": 0.017817148938775063, "reward_total_mean": 0.9767506122589111, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9767506122589111, "rewards/meter/std": 0.017817148938775063, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9767506122589111, "rewards/total_composite/std": 0.017817148938775063, "sampling/importance_sampling_ratio/max": 1.9086910486221313, "sampling/importance_sampling_ratio/mean": 1.0100852251052856, "sampling/importance_sampling_ratio/min": 0.09997859597206116, "sampling/sampling_logp_difference/max": 2.3027992248535156, "sampling/sampling_logp_difference/mean": 0.05213453993201256, "step": 2567 }, { "clip_ratio/high_max": 0.03215251048095524, "clip_ratio/high_mean": 0.03215251048095524, "clip_ratio/low_mean": 0.008893084479495883, "clip_ratio/low_min": 0.008893084479495883, "clip_ratio/region_mean": 0.041045594960451126, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 72.25, "completions/mean_terminated_length": 72.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.4912409372627735, "epoch": 0.1031449572237619, "frac_reward_zero_std": 0.0, "grad_norm": 4.675406455993652, "learning_rate": 2.2212121212121214e-06, "loss": -0.0085, "num_tokens": 5820304.0, "reward": 0.842863917350769, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.842863917350769, "reward_meter_std": 0.26622849702835083, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.26622846722602844, "reward_total_composite_mean": 0.842863917350769, "reward_total_composite_std": 0.26622849702835083, "reward_total_mean": 0.842863917350769, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.842863917350769, "rewards/meter/std": 0.26622849702835083, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.842863917350769, "rewards/total_composite/std": 0.26622849702835083, "sampling/importance_sampling_ratio/max": 1.7533190250396729, "sampling/importance_sampling_ratio/mean": 1.005039095878601, "sampling/importance_sampling_ratio/min": 0.2445313036441803, "sampling/sampling_logp_difference/max": 1.408411979675293, "sampling/sampling_logp_difference/mean": 0.06255535036325455, "step": 2568 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.002900286461226642, "epoch": 0.10318512270554685, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.2181818181818187e-06, "loss": 0.0, "num_tokens": 5821728.0, "reward": 0.9980231523513794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9980231523513794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0029308795928955, "sampling/importance_sampling_ratio/mean": 1.000272274017334, "sampling/importance_sampling_ratio/min": 0.9983823895454407, "sampling/sampling_logp_difference/max": 0.0029266122728586197, "sampling/sampling_logp_difference/mean": 0.0002904376306105405, "step": 2569 }, { "clip_ratio/high_max": 0.05010954383760691, "clip_ratio/high_mean": 0.05010954383760691, "clip_ratio/low_mean": 0.035002983175218105, "clip_ratio/low_min": 0.035002983175218105, "clip_ratio/region_mean": 0.08511252701282501, "completions/clipped_ratio": 0.0, "completions/max_length": 50.0, "completions/max_terminated_length": 50.0, "completions/mean_length": 43.0, "completions/mean_terminated_length": 43.0, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.562019981443882, "epoch": 0.1032252881873318, "frac_reward_zero_std": 0.0, "grad_norm": 12.921791076660156, "learning_rate": 2.2151515151515155e-06, "loss": 0.011, "num_tokens": 5823592.0, "reward": 0.8914755582809448, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8914755582809448, "reward_meter_std": 0.08936496078968048, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.08936495333909988, "reward_total_composite_mean": 0.8914755582809448, "reward_total_composite_std": 0.08936496078968048, "reward_total_mean": 0.8914755582809448, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8914755582809448, "rewards/meter/std": 0.08936496078968048, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8914755582809448, "rewards/total_composite/std": 0.08936496078968048, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9992160201072693, "sampling/importance_sampling_ratio/min": 0.15693165361881256, "sampling/sampling_logp_difference/max": 1.851944923400879, "sampling/sampling_logp_difference/mean": 0.09845032542943954, "step": 2570 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.0015226418327074498, "epoch": 0.10326545366911676, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.2121212121212124e-06, "loss": 0.0, "num_tokens": 5824992.0, "reward": 0.9943599104881287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9943599104881287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0017789602279663, "sampling/importance_sampling_ratio/mean": 1.0002009868621826, "sampling/importance_sampling_ratio/min": 1.0, "sampling/sampling_logp_difference/max": 0.0017774300649762154, "sampling/sampling_logp_difference/mean": 0.00020078742818441242, "step": 2571 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0033783784601837397, "clip_ratio/low_min": 0.0033783784601837397, "clip_ratio/region_mean": 0.0033783784601837397, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 37.0, "completions/mean_terminated_length": 37.0, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.054603309370577335, "epoch": 0.10330561915090171, "frac_reward_zero_std": 0.0, "grad_norm": 1.0375055074691772, "learning_rate": 2.209090909090909e-06, "loss": -0.0009, "num_tokens": 5826320.0, "reward": 0.9995728731155396, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9995728731155396, "reward_meter_std": 3.974059291067533e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.974912760895677e-05, "reward_total_composite_mean": 0.9995728731155396, "reward_total_composite_std": 3.974059291067533e-05, "reward_total_mean": 0.9995728731155396, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9995728731155396, "rewards/meter/std": 3.974059291067533e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995728731155396, "rewards/total_composite/std": 3.974059291067533e-05, "sampling/importance_sampling_ratio/max": 1.1907140016555786, "sampling/importance_sampling_ratio/mean": 1.0013068914413452, "sampling/importance_sampling_ratio/min": 0.5191892385482788, "sampling/sampling_logp_difference/max": 0.6554868221282959, "sampling/sampling_logp_difference/mean": 0.009617374278604984, "step": 2572 }, { "clip_ratio/high_max": 0.04813995969016105, "clip_ratio/high_mean": 0.04813995969016105, "clip_ratio/low_mean": 0.005514706019312143, "clip_ratio/low_min": 0.005514706019312143, "clip_ratio/region_mean": 0.05365466570947319, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 67.625, "completions/mean_terminated_length": 67.625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.5100602991878986, "epoch": 0.10334578463268666, "frac_reward_zero_std": 0.0, "grad_norm": 5.262681484222412, "learning_rate": 2.2060606060606064e-06, "loss": 0.0037, "num_tokens": 5828013.0, "reward": 0.8596113920211792, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9842870235443115, "reward_meter_std": 0.03939536586403847, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.34952226281166077, "reward_total_composite_mean": 0.8596113920211792, "reward_total_composite_std": 0.34952229261398315, "reward_total_mean": 0.8596113920211792, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9842870235443115, "rewards/meter/std": 0.03939536586403847, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8596113920211792, "rewards/total_composite/std": 0.34952229261398315, "sampling/importance_sampling_ratio/max": 1.7594947814941406, "sampling/importance_sampling_ratio/mean": 1.0002013444900513, "sampling/importance_sampling_ratio/min": 0.13531140983104706, "sampling/sampling_logp_difference/max": 2.000176429748535, "sampling/sampling_logp_difference/mean": 0.06369344890117645, "step": 2573 }, { "clip_ratio/high_max": 0.008656773250550032, "clip_ratio/high_mean": 0.008656773250550032, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/region_mean": 0.010417336598038673, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.25, "completions/mean_terminated_length": 72.25, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.09299392439424992, "epoch": 0.10338595011447162, "frac_reward_zero_std": 0.0, "grad_norm": 1.3739620447158813, "learning_rate": 2.2030303030303033e-06, "loss": -0.0004, "num_tokens": 5829911.0, "reward": 0.9990666508674622, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990666508674622, "reward_meter_std": 0.00011669577361317351, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011668946535792202, "reward_total_composite_mean": 0.9990666508674622, "reward_total_composite_std": 0.00011669577361317351, "reward_total_mean": 0.9990666508674622, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990666508674622, "rewards/meter/std": 0.00011669577361317351, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990666508674622, "rewards/total_composite/std": 0.00011669577361317351, "sampling/importance_sampling_ratio/max": 1.3253597021102905, "sampling/importance_sampling_ratio/mean": 1.0047719478607178, "sampling/importance_sampling_ratio/min": 0.3632344603538513, "sampling/sampling_logp_difference/max": 1.0127067565917969, "sampling/sampling_logp_difference/mean": 0.01342670526355505, "step": 2574 }, { "clip_ratio/high_max": 0.006622972548939288, "clip_ratio/high_mean": 0.006622972548939288, "clip_ratio/low_mean": 0.004734848625957966, "clip_ratio/low_min": 0.004734848625957966, "clip_ratio/region_mean": 0.011357821174897254, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 132.25, "completions/mean_terminated_length": 132.25, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.08119754772633314, "epoch": 0.10342611559625657, "frac_reward_zero_std": 0.0, "grad_norm": 0.6759248971939087, "learning_rate": 2.2e-06, "loss": -0.001, "num_tokens": 5832393.0, "reward": 0.9992468357086182, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992468357086182, "reward_meter_std": 8.659085870021954e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.659218292450532e-05, "reward_total_composite_mean": 0.9992468357086182, "reward_total_composite_std": 8.659085870021954e-05, "reward_total_mean": 0.9992468357086182, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992468357086182, "rewards/meter/std": 8.659085870021954e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992468357086182, "rewards/total_composite/std": 8.659085870021954e-05, "sampling/importance_sampling_ratio/max": 1.3609951734542847, "sampling/importance_sampling_ratio/mean": 1.0009084939956665, "sampling/importance_sampling_ratio/min": 0.2579735517501831, "sampling/sampling_logp_difference/max": 1.35489821434021, "sampling/sampling_logp_difference/mean": 0.010475593619048595, "step": 2575 }, { "clip_ratio/high_max": 0.014005606528371572, "clip_ratio/high_mean": 0.014005606528371572, "clip_ratio/low_mean": 0.0062509768176823854, "clip_ratio/low_min": 0.0062509768176823854, "clip_ratio/region_mean": 0.020256583346053958, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 80.375, "completions/mean_terminated_length": 80.375, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.21419258043169975, "epoch": 0.10346628107804152, "frac_reward_zero_std": 0.0, "grad_norm": 1.7544798851013184, "learning_rate": 2.196969696969697e-06, "loss": 0.0001, "num_tokens": 5834484.0, "reward": 0.9987189769744873, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987189769744873, "reward_meter_std": 0.00029780645854771137, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00029780055047012866, "reward_total_composite_mean": 0.9987189769744873, "reward_total_composite_std": 0.00029780645854771137, "reward_total_mean": 0.9987189769744873, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987189769744873, "rewards/meter/std": 0.00029780645854771137, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987189769744873, "rewards/total_composite/std": 0.00029780645854771137, "sampling/importance_sampling_ratio/max": 1.3658860921859741, "sampling/importance_sampling_ratio/mean": 1.005149483680725, "sampling/importance_sampling_ratio/min": 0.48345139622688293, "sampling/sampling_logp_difference/max": 0.7268044948577881, "sampling/sampling_logp_difference/mean": 0.023812301456928253, "step": 2576 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.015041560400277376, "epoch": 0.10350644655982648, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.193939393939394e-06, "loss": 0.0, "num_tokens": 5836412.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0703659057617188, "sampling/importance_sampling_ratio/mean": 1.0000942945480347, "sampling/importance_sampling_ratio/min": 0.882771372795105, "sampling/sampling_logp_difference/max": 0.12468904256820679, "sampling/sampling_logp_difference/mean": 0.0015179180772975087, "step": 2577 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0038007627008482814, "clip_ratio/low_min": 0.0038007627008482814, "clip_ratio/region_mean": 0.0038007627008482814, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 98.25, "completions/mean_terminated_length": 98.25, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.0589327453635633, "epoch": 0.10354661204161143, "frac_reward_zero_std": 0.0, "grad_norm": 0.25849685072898865, "learning_rate": 2.190909090909091e-06, "loss": 0.0004, "num_tokens": 5838518.0, "reward": 0.9980080127716064, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980080127716064, "reward_meter_std": 1.598181188455783e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.5980675016180612e-05, "reward_total_composite_mean": 0.9980080127716064, "reward_total_composite_std": 1.598181188455783e-05, "reward_total_mean": 0.9980080127716064, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980080127716064, "rewards/meter/std": 1.598181188455783e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980080127716064, "rewards/total_composite/std": 1.598181188455783e-05, "sampling/importance_sampling_ratio/max": 1.5758888721466064, "sampling/importance_sampling_ratio/mean": 1.0036529302597046, "sampling/importance_sampling_ratio/min": 0.3210567235946655, "sampling/sampling_logp_difference/max": 1.1361374855041504, "sampling/sampling_logp_difference/mean": 0.007937485352158546, "step": 2578 }, { "clip_ratio/high_max": 0.02378739172127098, "clip_ratio/high_mean": 0.02378739172127098, "clip_ratio/low_mean": 0.01786035276018083, "clip_ratio/low_min": 0.01786035276018083, "clip_ratio/region_mean": 0.04164774448145181, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 104.5, "completions/mean_terminated_length": 104.5, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.47100016474723816, "epoch": 0.10358677752339639, "frac_reward_zero_std": 0.0, "grad_norm": 3.5709714889526367, "learning_rate": 2.187878787878788e-06, "loss": 0.0013, "num_tokens": 5840674.0, "reward": 0.9814869165420532, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9814869165420532, "reward_meter_std": 0.012266860343515873, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.012266861274838448, "reward_total_composite_mean": 0.9814869165420532, "reward_total_composite_std": 0.012266860343515873, "reward_total_mean": 0.9814869165420532, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9814869165420532, "rewards/meter/std": 0.012266860343515873, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9814869165420532, "rewards/total_composite/std": 0.012266860343515873, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0061341524124146, "sampling/importance_sampling_ratio/min": 0.15966860949993134, "sampling/sampling_logp_difference/max": 1.8346548080444336, "sampling/sampling_logp_difference/mean": 0.05761587247252464, "step": 2579 }, { "clip_ratio/high_max": 0.01871931995265186, "clip_ratio/high_mean": 0.01871931995265186, "clip_ratio/low_mean": 0.008511104388162494, "clip_ratio/low_min": 0.008511104388162494, "clip_ratio/region_mean": 0.027230424340814352, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 59.0, "completions/mean_terminated_length": 59.0, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.25036873295903206, "epoch": 0.10362694300518134, "frac_reward_zero_std": 0.0, "grad_norm": 4.734373569488525, "learning_rate": 2.184848484848485e-06, "loss": 0.009, "num_tokens": 5842442.0, "reward": 0.9932022094726562, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9932022094726562, "reward_meter_std": 0.002190361265093088, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002190381521359086, "reward_total_composite_mean": 0.9932022094726562, "reward_total_composite_std": 0.002190361265093088, "reward_total_mean": 0.9932022094726562, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9932022094726562, "rewards/meter/std": 0.002190361265093088, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9932022094726562, "rewards/total_composite/std": 0.002190361265093088, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0038007497787476, "sampling/importance_sampling_ratio/min": 0.2926851809024811, "sampling/sampling_logp_difference/max": 1.2286577224731445, "sampling/sampling_logp_difference/mean": 0.03438284620642662, "step": 2580 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.006089928210712969, "epoch": 0.1036671084869663, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.181818181818182e-06, "loss": 0.0, "num_tokens": 5844282.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0165749788284302, "sampling/importance_sampling_ratio/mean": 0.9999182224273682, "sampling/importance_sampling_ratio/min": 0.9637494087219238, "sampling/sampling_logp_difference/max": 0.03692399710416794, "sampling/sampling_logp_difference/mean": 0.0005618593422695994, "step": 2581 }, { "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/low_mean": 0.0017123287543654442, "clip_ratio/low_min": 0.0017123287543654442, "clip_ratio/region_mean": 0.003448439878411591, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.875, "completions/mean_terminated_length": 72.875, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.0825590007007122, "epoch": 0.10370727396875125, "frac_reward_zero_std": 0.0, "grad_norm": 1.6909281015396118, "learning_rate": 2.1787878787878788e-06, "loss": -0.0011, "num_tokens": 5846329.0, "reward": 0.9989848136901855, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989848136901855, "reward_meter_std": 0.00030317477649077773, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0003031773376278579, "reward_total_composite_mean": 0.9989848136901855, "reward_total_composite_std": 0.00030317477649077773, "reward_total_mean": 0.9989848136901855, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989848136901855, "rewards/meter/std": 0.00030317477649077773, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989848136901855, "rewards/total_composite/std": 0.00030317477649077773, "sampling/importance_sampling_ratio/max": 1.3127074241638184, "sampling/importance_sampling_ratio/mean": 1.0025291442871094, "sampling/importance_sampling_ratio/min": 0.2867645025253296, "sampling/sampling_logp_difference/max": 1.249094009399414, "sampling/sampling_logp_difference/mean": 0.014997649937868118, "step": 2582 }, { "clip_ratio/high_max": 0.057334170676767826, "clip_ratio/high_mean": 0.057334170676767826, "clip_ratio/low_mean": 0.005081300623714924, "clip_ratio/low_min": 0.005081300623714924, "clip_ratio/region_mean": 0.06241547130048275, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 120.375, "completions/mean_terminated_length": 120.375, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.4293239116668701, "epoch": 0.1037474394505362, "frac_reward_zero_std": 0.0, "grad_norm": 4.086141109466553, "learning_rate": 2.175757575757576e-06, "loss": 0.0175, "num_tokens": 5848580.0, "reward": 0.9723613262176514, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.989915132522583, "reward_meter_std": 0.006506068166345358, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05276636406779289, "reward_total_composite_mean": 0.9723613262176514, "reward_total_composite_std": 0.052766378968954086, "reward_total_mean": 0.9723613262176514, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.989915132522583, "rewards/meter/std": 0.006506068166345358, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9723613262176514, "rewards/total_composite/std": 0.052766378968954086, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9998201727867126, "sampling/importance_sampling_ratio/min": 0.21890626847743988, "sampling/sampling_logp_difference/max": 1.5191116333007812, "sampling/sampling_logp_difference/mean": 0.059543609619140625, "step": 2583 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0018527817592257634, "epoch": 0.10378760493232116, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.172727272727273e-06, "loss": 0.0, "num_tokens": 5850268.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0029222965240479, "sampling/importance_sampling_ratio/mean": 1.0002394914627075, "sampling/importance_sampling_ratio/min": 0.9997970461845398, "sampling/sampling_logp_difference/max": 0.002918010577559471, "sampling/sampling_logp_difference/mean": 0.00024031523207668215, "step": 2584 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0019878629682352766, "epoch": 0.10382777041410611, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.16969696969697e-06, "loss": 0.0, "num_tokens": 5852236.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0034657716751099, "sampling/importance_sampling_ratio/mean": 1.0002455711364746, "sampling/importance_sampling_ratio/min": 0.9998869299888611, "sampling/sampling_logp_difference/max": 0.0034597045741975307, "sampling/sampling_logp_difference/mean": 0.0002460446848999709, "step": 2585 }, { "clip_ratio/high_max": 0.012192239752039313, "clip_ratio/high_mean": 0.012192239752039313, "clip_ratio/low_mean": 0.01918130088597536, "clip_ratio/low_min": 0.01918130088597536, "clip_ratio/region_mean": 0.031373540638014674, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 365.5, "completions/mean_terminated_length": 365.5, "completions/min_length": 342.0, "completions/min_terminated_length": 342.0, "entropy": 0.4673043563961983, "epoch": 0.10386793589589108, "frac_reward_zero_std": 0.0, "grad_norm": 1.9402741193771362, "learning_rate": 2.166666666666667e-06, "loss": -0.0205, "num_tokens": 5856712.0, "reward": 0.7631400227546692, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7916666269302368, "reward_count_adherence_std": 0.04454353079199791, "reward_meter_mean": 0.9985870122909546, "reward_meter_std": 0.000464948097942397, "reward_repeat_penalty_mean": 0.9663312435150146, "reward_repeat_penalty_std": 0.039662934839725494, "reward_std": 0.03845897316932678, "reward_total_composite_mean": 0.7631400227546692, "reward_total_composite_std": 0.03845896199345589, "reward_total_mean": 0.7631400227546692, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7916666269302368, "rewards/count_adherence/std": 0.04454353079199791, "rewards/meter/mean": 0.9985870122909546, "rewards/meter/std": 0.000464948097942397, "rewards/repeat_penalty/mean": 0.9663312435150146, "rewards/repeat_penalty/std": 0.039662934839725494, "rewards/total_composite/mean": 0.7631400227546692, "rewards/total_composite/std": 0.03845896199345589, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0101393461227417, "sampling/importance_sampling_ratio/min": 0.07378236204385757, "sampling/sampling_logp_difference/max": 2.606635570526123, "sampling/sampling_logp_difference/mean": 0.05639910697937012, "step": 2586 }, { "clip_ratio/high_max": 0.038169488427229226, "clip_ratio/high_mean": 0.038169488427229226, "clip_ratio/low_mean": 0.005154639016836882, "clip_ratio/low_min": 0.005154639016836882, "clip_ratio/region_mean": 0.04332412744406611, "completions/clipped_ratio": 0.0, "completions/max_length": 106.0, "completions/max_terminated_length": 106.0, "completions/mean_length": 100.875, "completions/mean_terminated_length": 100.875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.4231472760438919, "epoch": 0.10390810137767603, "frac_reward_zero_std": 0.0, "grad_norm": 6.472557067871094, "learning_rate": 2.163636363636364e-06, "loss": -0.0067, "num_tokens": 5858927.0, "reward": 0.979087233543396, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.979087233543396, "reward_meter_std": 0.056263912469148636, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.056263916194438934, "reward_total_composite_mean": 0.979087233543396, "reward_total_composite_std": 0.056263912469148636, "reward_total_mean": 0.979087233543396, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.979087233543396, "rewards/meter/std": 0.056263912469148636, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.979087233543396, "rewards/total_composite/std": 0.056263912469148636, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0070948600769043, "sampling/importance_sampling_ratio/min": 0.023454248905181885, "sampling/sampling_logp_difference/max": 3.7527036666870117, "sampling/sampling_logp_difference/mean": 0.04917147755622864, "step": 2587 }, { "clip_ratio/high_max": 0.019755184242967516, "clip_ratio/high_mean": 0.019755184242967516, "clip_ratio/low_mean": 0.0030577474972233176, "clip_ratio/low_min": 0.0030577474972233176, "clip_ratio/region_mean": 0.022812931740190834, "completions/clipped_ratio": 0.0, "completions/max_length": 205.0, "completions/max_terminated_length": 205.0, "completions/mean_length": 202.5, "completions/mean_terminated_length": 202.5, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "entropy": 0.16349555738270283, "epoch": 0.10394826685946099, "frac_reward_zero_std": 0.0, "grad_norm": 1.698296308517456, "learning_rate": 2.1606060606060606e-06, "loss": 0.0085, "num_tokens": 5862067.0, "reward": 0.9727857112884521, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9954801797866821, "reward_meter_std": 0.008464212529361248, "reward_repeat_penalty_mean": 0.9772727489471436, "reward_repeat_penalty_std": 0.04208271950483322, "reward_std": 0.04096178337931633, "reward_total_composite_mean": 0.9727857112884521, "reward_total_composite_std": 0.04096178710460663, "reward_total_mean": 0.9727857112884521, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9954801797866821, "rewards/meter/std": 0.008464212529361248, "rewards/repeat_penalty/mean": 0.9772727489471436, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.9727857112884521, "rewards/total_composite/std": 0.04096178710460663, "sampling/importance_sampling_ratio/max": 1.8658126592636108, "sampling/importance_sampling_ratio/mean": 1.0026134252548218, "sampling/importance_sampling_ratio/min": 0.23083457350730896, "sampling/sampling_logp_difference/max": 1.4660539627075195, "sampling/sampling_logp_difference/mean": 0.02448193170130253, "step": 2588 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0016017362504499033, "epoch": 0.10398843234124594, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.157575757575758e-06, "loss": 0.0, "num_tokens": 5863619.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0021939277648926, "sampling/importance_sampling_ratio/mean": 1.0002191066741943, "sampling/importance_sampling_ratio/min": 0.9999104142189026, "sampling/sampling_logp_difference/max": 0.0021914218086749315, "sampling/sampling_logp_difference/mean": 0.00021978352742735296, "step": 2589 }, { "clip_ratio/high_max": 0.03218326787464321, "clip_ratio/high_mean": 0.03218326787464321, "clip_ratio/low_mean": 0.013986697886139154, "clip_ratio/low_min": 0.013986697886139154, "clip_ratio/region_mean": 0.04616996576078236, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 70.5, "completions/mean_terminated_length": 70.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.514446884393692, "epoch": 0.1040285978230309, "frac_reward_zero_std": 0.0, "grad_norm": 6.6089911460876465, "learning_rate": 2.1545454545454547e-06, "loss": 0.0105, "num_tokens": 5865559.0, "reward": 0.9882650971412659, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9882650971412659, "reward_meter_std": 0.010569445788860321, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010569446720182896, "reward_total_composite_mean": 0.9882650971412659, "reward_total_composite_std": 0.010569445788860321, "reward_total_mean": 0.9882650971412659, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9882650971412659, "rewards/meter/std": 0.010569445788860321, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9882650971412659, "rewards/total_composite/std": 0.010569445788860321, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0161855220794678, "sampling/importance_sampling_ratio/min": 0.12965509295463562, "sampling/sampling_logp_difference/max": 2.042877435684204, "sampling/sampling_logp_difference/mean": 0.05746662616729736, "step": 2590 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.002733113680733368, "epoch": 0.10406876330481585, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.1515151515151515e-06, "loss": 0.0, "num_tokens": 5866935.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0062510967254639, "sampling/importance_sampling_ratio/mean": 1.0003573894500732, "sampling/importance_sampling_ratio/min": 0.9997910261154175, "sampling/sampling_logp_difference/max": 0.0062316060066223145, "sampling/sampling_logp_difference/mean": 0.0003589413536246866, "step": 2591 }, { "clip_ratio/high_max": 0.019220190355554223, "clip_ratio/high_mean": 0.019220190355554223, "clip_ratio/low_mean": 0.014869472477585077, "clip_ratio/low_min": 0.014869472477585077, "clip_ratio/region_mean": 0.0340896628331393, "completions/clipped_ratio": 0.0, "completions/max_length": 60.0, "completions/max_terminated_length": 60.0, "completions/mean_length": 59.0, "completions/mean_terminated_length": 59.0, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.2373839858919382, "epoch": 0.1041089287866008, "frac_reward_zero_std": 0.0, "grad_norm": 5.403855800628662, "learning_rate": 2.148484848484849e-06, "loss": 0.0079, "num_tokens": 5868615.0, "reward": 0.9917755126953125, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9917755126953125, "reward_meter_std": 0.003381570801138878, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0033815728966146708, "reward_total_composite_mean": 0.9917755126953125, "reward_total_composite_std": 0.003381570801138878, "reward_total_mean": 0.9917755126953125, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9917755126953125, "rewards/meter/std": 0.003381570801138878, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917755126953125, "rewards/total_composite/std": 0.003381570801138878, "sampling/importance_sampling_ratio/max": 1.8206579685211182, "sampling/importance_sampling_ratio/mean": 1.003715991973877, "sampling/importance_sampling_ratio/min": 0.3259781301021576, "sampling/sampling_logp_difference/max": 1.120924949645996, "sampling/sampling_logp_difference/mean": 0.03163475543260574, "step": 2592 }, { "clip_ratio/high_max": 0.03610773291438818, "clip_ratio/high_mean": 0.03610773291438818, "clip_ratio/low_mean": 0.0256543830037117, "clip_ratio/low_min": 0.0256543830037117, "clip_ratio/region_mean": 0.06176211591809988, "completions/clipped_ratio": 0.0, "completions/max_length": 239.0, "completions/max_terminated_length": 239.0, "completions/mean_length": 216.375, "completions/mean_terminated_length": 216.375, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "entropy": 0.5770430602133274, "epoch": 0.10414909426838576, "frac_reward_zero_std": 0.0, "grad_norm": 3.471240282058716, "learning_rate": 2.1454545454545456e-06, "loss": 0.0705, "num_tokens": 5871770.0, "reward": 0.921344518661499, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9464285969734192, "reward_count_adherence_std": 0.07393559068441391, "reward_meter_mean": 0.990897536277771, "reward_meter_std": 0.006994582246989012, "reward_repeat_penalty_mean": 0.9820512533187866, "reward_repeat_penalty_std": 0.033347416669130325, "reward_std": 0.0833897814154625, "reward_total_composite_mean": 0.921344518661499, "reward_total_composite_std": 0.0833897739648819, "reward_total_mean": 0.921344518661499, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9464285969734192, "rewards/count_adherence/std": 0.07393559068441391, "rewards/meter/mean": 0.990897536277771, "rewards/meter/std": 0.006994582246989012, "rewards/repeat_penalty/mean": 0.9820512533187866, "rewards/repeat_penalty/std": 0.033347416669130325, "rewards/total_composite/mean": 0.921344518661499, "rewards/total_composite/std": 0.0833897739648819, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0076571702957153, "sampling/importance_sampling_ratio/min": 0.16753454506397247, "sampling/sampling_logp_difference/max": 1.7865657806396484, "sampling/sampling_logp_difference/mean": 0.07251539081335068, "step": 2593 }, { "clip_ratio/high_max": 0.025721082463860512, "clip_ratio/high_mean": 0.025721082463860512, "clip_ratio/low_mean": 0.010563877876847982, "clip_ratio/low_min": 0.010563877876847982, "clip_ratio/region_mean": 0.036284960340708494, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 58.875, "completions/mean_terminated_length": 58.875, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.27704715728759766, "epoch": 0.10418925975017071, "frac_reward_zero_std": 0.0, "grad_norm": 5.423774719238281, "learning_rate": 2.1424242424242425e-06, "loss": 0.0196, "num_tokens": 5873489.0, "reward": 0.992070198059082, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.992070198059082, "reward_meter_std": 0.0043101259507238865, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004310120362788439, "reward_total_composite_mean": 0.992070198059082, "reward_total_composite_std": 0.0043101259507238865, "reward_total_mean": 0.992070198059082, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.992070198059082, "rewards/meter/std": 0.0043101259507238865, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992070198059082, "rewards/total_composite/std": 0.0043101259507238865, "sampling/importance_sampling_ratio/max": 1.3894490003585815, "sampling/importance_sampling_ratio/mean": 0.9979619383811951, "sampling/importance_sampling_ratio/min": 0.24702267348766327, "sampling/sampling_logp_difference/max": 1.3982751369476318, "sampling/sampling_logp_difference/mean": 0.03451910987496376, "step": 2594 }, { "clip_ratio/high_max": 0.012042732443660498, "clip_ratio/high_mean": 0.012042732443660498, "clip_ratio/low_mean": 0.0235387550201267, "clip_ratio/low_min": 0.0235387550201267, "clip_ratio/region_mean": 0.0355814874637872, "completions/clipped_ratio": 0.0, "completions/max_length": 307.0, "completions/max_terminated_length": 307.0, "completions/mean_length": 299.5, "completions/mean_terminated_length": 299.5, "completions/min_length": 293.0, "completions/min_terminated_length": 293.0, "entropy": 0.3467197176069021, "epoch": 0.10422942523195566, "frac_reward_zero_std": 0.0, "grad_norm": 2.122164726257324, "learning_rate": 2.1393939393939393e-06, "loss": -0.0043, "num_tokens": 5877533.0, "reward": 0.8223901987075806, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9690572023391724, "reward_meter_std": 0.0723068043589592, "reward_repeat_penalty_mean": 0.970588207244873, "reward_repeat_penalty_std": 0.03144249692559242, "reward_std": 0.05969364941120148, "reward_total_composite_mean": 0.8223901987075806, "reward_total_composite_std": 0.05969366803765297, "reward_total_mean": 0.8223901987075806, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9690572023391724, "rewards/meter/std": 0.0723068043589592, "rewards/repeat_penalty/mean": 0.970588207244873, "rewards/repeat_penalty/std": 0.03144249692559242, "rewards/total_composite/mean": 0.8223901987075806, "rewards/total_composite/std": 0.05969366803765297, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0036664009094238, "sampling/importance_sampling_ratio/min": 0.1707945466041565, "sampling/sampling_logp_difference/max": 1.767293930053711, "sampling/sampling_logp_difference/mean": 0.04698567092418671, "step": 2595 }, { "clip_ratio/high_max": 0.01985564432106912, "clip_ratio/high_mean": 0.01985564432106912, "clip_ratio/low_mean": 0.005782836233265698, "clip_ratio/low_min": 0.005782836233265698, "clip_ratio/region_mean": 0.02563848055433482, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 510.0, "completions/mean_length": 501.125, "completions/mean_terminated_length": 499.5714416503906, "completions/min_length": 470.0, "completions/min_terminated_length": 470.0, "entropy": 0.34851858019828796, "epoch": 0.10426959071374062, "frac_reward_zero_std": 0.0, "grad_norm": 1.1440000534057617, "learning_rate": 2.1363636363636365e-06, "loss": 0.112, "num_tokens": 5882790.0, "reward": 0.7341695427894592, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8046875, "reward_count_adherence_std": 0.022097086533904076, "reward_meter_mean": 0.9981234073638916, "reward_meter_std": 0.0008658914593979716, "reward_repeat_penalty_mean": 0.9128260612487793, "reward_repeat_penalty_std": 0.08217112720012665, "reward_std": 0.07871916890144348, "reward_total_composite_mean": 0.7341695427894592, "reward_total_composite_std": 0.07871917635202408, "reward_total_mean": 0.7341695427894592, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8046875, "rewards/count_adherence/std": 0.022097086533904076, "rewards/meter/mean": 0.9981234073638916, "rewards/meter/std": 0.0008658914593979716, "rewards/repeat_penalty/mean": 0.9128260612487793, "rewards/repeat_penalty/std": 0.08217112720012665, "rewards/total_composite/mean": 0.7341695427894592, "rewards/total_composite/std": 0.07871917635202408, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.011509895324707, "sampling/importance_sampling_ratio/min": 0.19811558723449707, "sampling/sampling_logp_difference/max": 1.6189045906066895, "sampling/sampling_logp_difference/mean": 0.048293329775333405, "step": 2596 }, { "clip_ratio/high_max": 0.010750473011285067, "clip_ratio/high_mean": 0.010750473011285067, "clip_ratio/low_mean": 0.01861410157289356, "clip_ratio/low_min": 0.01861410157289356, "clip_ratio/region_mean": 0.029364574584178627, "completions/clipped_ratio": 0.0, "completions/max_length": 135.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 122.625, "completions/mean_terminated_length": 122.625, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.3435132773593068, "epoch": 0.10430975619552557, "frac_reward_zero_std": 0.0, "grad_norm": 3.617664098739624, "learning_rate": 2.133333333333334e-06, "loss": -0.0172, "num_tokens": 5884995.0, "reward": 0.8836444616317749, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9898524284362793, "reward_meter_std": 0.005112205166369677, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.0630328580737114, "reward_total_composite_mean": 0.8836444616317749, "reward_total_composite_std": 0.06303286552429199, "reward_total_mean": 0.8836444616317749, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9898524284362793, "rewards/meter/std": 0.005112205166369677, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8836444616317749, "rewards/total_composite/std": 0.06303286552429199, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.013502597808838, "sampling/importance_sampling_ratio/min": 0.32310131192207336, "sampling/sampling_logp_difference/max": 1.1297893524169922, "sampling/sampling_logp_difference/mean": 0.034280113875865936, "step": 2597 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 28.875, "completions/mean_terminated_length": 28.875, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.006841129012173042, "epoch": 0.10434992167731053, "frac_reward_zero_std": 0.0, "grad_norm": 7.509369850158691, "learning_rate": 2.1303030303030306e-06, "loss": -0.0118, "num_tokens": 5886546.0, "reward": 0.9953750371932983, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953750371932983, "reward_meter_std": 0.0010425866348668933, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010425925720483065, "reward_total_composite_mean": 0.9953750371932983, "reward_total_composite_std": 0.0010425866348668933, "reward_total_mean": 0.9953750371932983, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953750371932983, "rewards/meter/std": 0.0010425866348668933, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953750371932983, "rewards/total_composite/std": 0.0010425866348668933, "sampling/importance_sampling_ratio/max": 1.1243258714675903, "sampling/importance_sampling_ratio/mean": 0.9976253509521484, "sampling/importance_sampling_ratio/min": 0.35399580001831055, "sampling/sampling_logp_difference/max": 1.0384702682495117, "sampling/sampling_logp_difference/mean": 0.00547828758135438, "step": 2598 }, { "clip_ratio/high_max": 0.007614409434609115, "clip_ratio/high_mean": 0.007614409434609115, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007614409434609115, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.06834368547424674, "epoch": 0.10439008715909548, "frac_reward_zero_std": 0.0, "grad_norm": 1.446718692779541, "learning_rate": 2.1272727272727275e-06, "loss": 0.0041, "num_tokens": 5888682.0, "reward": 0.9992596507072449, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992596507072449, "reward_meter_std": 0.0001470481656724587, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000147052516695112, "reward_total_composite_mean": 0.9992596507072449, "reward_total_composite_std": 0.0001470481656724587, "reward_total_mean": 0.9992596507072449, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992596507072449, "rewards/meter/std": 0.0001470481656724587, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992596507072449, "rewards/total_composite/std": 0.0001470481656724587, "sampling/importance_sampling_ratio/max": 1.4146671295166016, "sampling/importance_sampling_ratio/mean": 0.9985703229904175, "sampling/importance_sampling_ratio/min": 0.3790672719478607, "sampling/sampling_logp_difference/max": 0.9700416326522827, "sampling/sampling_logp_difference/mean": 0.011022298596799374, "step": 2599 }, { "clip_ratio/high_max": 0.007352941203862429, "clip_ratio/high_mean": 0.007352941203862429, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007352941203862429, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.5, "completions/mean_terminated_length": 34.5, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.1272981008514762, "epoch": 0.10443025264088043, "frac_reward_zero_std": 0.0, "grad_norm": 5.452374458312988, "learning_rate": 2.1242424242424243e-06, "loss": 0.0094, "num_tokens": 5890094.0, "reward": 0.9899568557739258, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9899568557739258, "reward_meter_std": 0.008095722645521164, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00809571985155344, "reward_total_composite_mean": 0.9899568557739258, "reward_total_composite_std": 0.008095722645521164, "reward_total_mean": 0.9899568557739258, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9899568557739258, "rewards/meter/std": 0.008095722645521164, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9899568557739258, "rewards/total_composite/std": 0.008095722645521164, "sampling/importance_sampling_ratio/max": 1.2015459537506104, "sampling/importance_sampling_ratio/mean": 0.9940556287765503, "sampling/importance_sampling_ratio/min": 0.19894060492515564, "sampling/sampling_logp_difference/max": 1.6147489547729492, "sampling/sampling_logp_difference/mean": 0.02594204805791378, "step": 2600 }, { "epoch": 0.10443025264088043, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/max_length": 414.3076923076923, "eval_completions/max_terminated_length": 392.0, "eval_completions/mean_length": 209.5096153846154, "eval_completions/mean_terminated_length": 203.01236314039963, "eval_completions/min_length": 60.92307692307692, "eval_completions/min_terminated_length": 60.92307692307692, "eval_entropy": 0.40257097207582915, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 5890094.0, "eval_reward": 0.6867790864064143, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9526959611819341, "eval_reward_count_adherence_std": 0.0694849301989262, "eval_reward_meter_mean": 0.7689516819440402, "eval_reward_meter_std": 0.3567994758486748, "eval_reward_repeat_penalty_mean": 0.92968055835137, "eval_reward_repeat_penalty_std": 0.09778400797110337, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6867790864064143, "eval_reward_total_composite_std": 0.3467414522400269, "eval_reward_total_mean": 0.6867790864064143, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9526959611819341, "eval_rewards/count_adherence/std": 0.0694849301989262, "eval_rewards/meter/mean": 0.7689516819440402, "eval_rewards/meter/std": 0.3567994758486748, "eval_rewards/repeat_penalty/mean": 0.92968055835137, "eval_rewards/repeat_penalty/std": 0.09778400797110337, "eval_rewards/total_composite/mean": 0.6867790864064143, "eval_rewards/total_composite/std": 0.3467414522400269, "eval_runtime": 77.3652, "eval_samples_per_second": 1.344, "eval_sampling/importance_sampling_ratio/max": 1.5790529526196992, "eval_sampling/importance_sampling_ratio/mean": 1.0097874861497145, "eval_sampling/importance_sampling_ratio/min": 0.3349975932102937, "eval_sampling/sampling_logp_difference/max": 1.1111029111422026, "eval_sampling/sampling_logp_difference/mean": 0.03625713331768146, "eval_steps_per_second": 0.168, "step": 2600 }, { "clip_ratio/high_max": 0.0017123287543654442, "clip_ratio/high_mean": 0.0017123287543654442, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0017123287543654442, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.125, "completions/mean_terminated_length": 72.125, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.06924279127269983, "epoch": 0.10447041812266539, "frac_reward_zero_std": 0.0, "grad_norm": 1.6252681016921997, "learning_rate": 2.1212121212121216e-06, "loss": -0.0003, "num_tokens": 5891951.0, "reward": 0.9991148710250854, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991148710250854, "reward_meter_std": 0.00010878306056838483, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010877339082071558, "reward_total_composite_mean": 0.9991148710250854, "reward_total_composite_std": 0.00010878306056838483, "reward_total_mean": 0.9991148710250854, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991148710250854, "rewards/meter/std": 0.00010878306056838483, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991148710250854, "rewards/total_composite/std": 0.00010878306056838483, "sampling/importance_sampling_ratio/max": 1.4661310911178589, "sampling/importance_sampling_ratio/mean": 1.0020719766616821, "sampling/importance_sampling_ratio/min": 0.3458172678947449, "sampling/sampling_logp_difference/max": 1.061844825744629, "sampling/sampling_logp_difference/mean": 0.010965629480779171, "step": 2601 }, { "clip_ratio/high_max": 0.020244444953277707, "clip_ratio/high_mean": 0.020244444953277707, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.022138384403660893, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.24536396749317646, "epoch": 0.10451058360445034, "frac_reward_zero_std": 0.0, "grad_norm": 8.062291145324707, "learning_rate": 2.1181818181818184e-06, "loss": -0.0044, "num_tokens": 5893679.0, "reward": 0.9672091007232666, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9672091007232666, "reward_meter_std": 0.05921924114227295, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.059219252318143845, "reward_total_composite_mean": 0.9672091007232666, "reward_total_composite_std": 0.05921924114227295, "reward_total_mean": 0.9672091007232666, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9672091007232666, "rewards/meter/std": 0.05921924114227295, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9672091007232666, "rewards/total_composite/std": 0.05921924114227295, "sampling/importance_sampling_ratio/max": 1.7304493188858032, "sampling/importance_sampling_ratio/mean": 1.0095655918121338, "sampling/importance_sampling_ratio/min": 0.444327712059021, "sampling/sampling_logp_difference/max": 0.8111929893493652, "sampling/sampling_logp_difference/mean": 0.03210901468992233, "step": 2602 }, { "clip_ratio/high_max": 0.0017123287543654442, "clip_ratio/high_mean": 0.0017123287543654442, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/region_mean": 0.0051369862630963326, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 72.375, "completions/mean_terminated_length": 72.375, "completions/min_length": 72.0, "completions/min_terminated_length": 72.0, "entropy": 0.05477497586980462, "epoch": 0.1045507490862353, "frac_reward_zero_std": 0.0, "grad_norm": 0.387520432472229, "learning_rate": 2.1151515151515152e-06, "loss": -0.002, "num_tokens": 5895490.0, "reward": 0.9991236925125122, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991236925125122, "reward_meter_std": 0.00013291009236127138, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00013291911454871297, "reward_total_composite_mean": 0.9991236925125122, "reward_total_composite_std": 0.00013291009236127138, "reward_total_mean": 0.9991236925125122, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991236925125122, "rewards/meter/std": 0.00013291009236127138, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991236925125122, "rewards/total_composite/std": 0.00013291009236127138, "sampling/importance_sampling_ratio/max": 1.3512102365493774, "sampling/importance_sampling_ratio/mean": 1.0055519342422485, "sampling/importance_sampling_ratio/min": 0.8066705465316772, "sampling/sampling_logp_difference/max": 0.30100059509277344, "sampling/sampling_logp_difference/mean": 0.00656279968097806, "step": 2603 }, { "clip_ratio/high_max": 0.007380377617664635, "clip_ratio/high_mean": 0.007380377617664635, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/region_mean": 0.010852599865756929, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.07368437433615327, "epoch": 0.10459091456802025, "frac_reward_zero_std": 0.0, "grad_norm": 4.805464267730713, "learning_rate": 2.1121212121212125e-06, "loss": 0.0218, "num_tokens": 5897338.0, "reward": 0.9956408739089966, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956408739089966, "reward_meter_std": 0.004533765371888876, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0045337737537920475, "reward_total_composite_mean": 0.9956408739089966, "reward_total_composite_std": 0.004533765371888876, "reward_total_mean": 0.9956408739089966, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956408739089966, "rewards/meter/std": 0.004533765371888876, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9956408739089966, "rewards/total_composite/std": 0.004533765371888876, "sampling/importance_sampling_ratio/max": 1.9683966636657715, "sampling/importance_sampling_ratio/mean": 1.0015579462051392, "sampling/importance_sampling_ratio/min": 0.2769683301448822, "sampling/sampling_logp_difference/max": 1.2838521003723145, "sampling/sampling_logp_difference/mean": 0.015114509500563145, "step": 2604 }, { "clip_ratio/high_max": 0.018881317228078842, "clip_ratio/high_mean": 0.018881317228078842, "clip_ratio/low_mean": 0.00836864416487515, "clip_ratio/low_min": 0.00836864416487515, "clip_ratio/region_mean": 0.027249961392953992, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 59.375, "completions/mean_terminated_length": 59.375, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.18665459845215082, "epoch": 0.1046310800498052, "frac_reward_zero_std": 0.0, "grad_norm": 3.96225905418396, "learning_rate": 2.1090909090909093e-06, "loss": -0.0034, "num_tokens": 5899077.0, "reward": 0.9955159425735474, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955159425735474, "reward_meter_std": 0.0011379423085600138, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001137925311923027, "reward_total_composite_mean": 0.9955159425735474, "reward_total_composite_std": 0.0011379423085600138, "reward_total_mean": 0.9955159425735474, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955159425735474, "rewards/meter/std": 0.0011379423085600138, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955159425735474, "rewards/total_composite/std": 0.0011379423085600138, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0065569877624512, "sampling/importance_sampling_ratio/min": 0.45299965143203735, "sampling/sampling_logp_difference/max": 0.7918639183044434, "sampling/sampling_logp_difference/mean": 0.022009698674082756, "step": 2605 }, { "clip_ratio/high_max": 0.04019012441858649, "clip_ratio/high_mean": 0.04019012441858649, "clip_ratio/low_mean": 0.010766045656055212, "clip_ratio/low_min": 0.010766045656055212, "clip_ratio/region_mean": 0.050956170074641705, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 137.25, "completions/mean_terminated_length": 137.25, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.44699474796652794, "epoch": 0.10467124553159016, "frac_reward_zero_std": 0.0, "grad_norm": 3.778212070465088, "learning_rate": 2.106060606060606e-06, "loss": 0.015, "num_tokens": 5901663.0, "reward": 0.997505784034729, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997505784034729, "reward_meter_std": 0.002905552973970771, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0029055566992610693, "reward_total_composite_mean": 0.997505784034729, "reward_total_composite_std": 0.002905552973970771, "reward_total_mean": 0.997505784034729, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997505784034729, "rewards/meter/std": 0.002905552973970771, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997505784034729, "rewards/total_composite/std": 0.002905552973970771, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0051594972610474, "sampling/importance_sampling_ratio/min": 0.30966120958328247, "sampling/sampling_logp_difference/max": 1.172276496887207, "sampling/sampling_logp_difference/mean": 0.05034590885043144, "step": 2606 }, { "clip_ratio/high_max": 0.023096036864444613, "clip_ratio/high_mean": 0.023096036864444613, "clip_ratio/low_mean": 0.0025254478096030653, "clip_ratio/low_min": 0.0025254478096030653, "clip_ratio/region_mean": 0.02562148467404768, "completions/clipped_ratio": 0.0, "completions/max_length": 199.0, "completions/max_terminated_length": 199.0, "completions/mean_length": 195.875, "completions/mean_terminated_length": 195.875, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.3686945028603077, "epoch": 0.10471141101337511, "frac_reward_zero_std": 0.0, "grad_norm": 2.1808922290802, "learning_rate": 2.103030303030303e-06, "loss": 0.0094, "num_tokens": 5904846.0, "reward": 0.9711034893989563, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988412261009216, "reward_meter_std": 0.0004946960834786296, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05153718218207359, "reward_total_composite_mean": 0.9711034893989563, "reward_total_composite_std": 0.05153718218207359, "reward_total_mean": 0.9711034893989563, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988412261009216, "rewards/meter/std": 0.0004946960834786296, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9711034893989563, "rewards/total_composite/std": 0.05153718218207359, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0106346607208252, "sampling/importance_sampling_ratio/min": 0.18494117259979248, "sampling/sampling_logp_difference/max": 1.6877174377441406, "sampling/sampling_logp_difference/mean": 0.04786364734172821, "step": 2607 }, { "clip_ratio/high_max": 0.020967748598195612, "clip_ratio/high_mean": 0.020967748598195612, "clip_ratio/low_mean": 0.0021739129442721605, "clip_ratio/low_min": 0.0021739129442721605, "clip_ratio/region_mean": 0.023141661542467773, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 118.625, "completions/mean_terminated_length": 118.625, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.283309917896986, "epoch": 0.10475157649516006, "frac_reward_zero_std": 0.0, "grad_norm": 2.9761064052581787, "learning_rate": 2.1000000000000002e-06, "loss": -0.0114, "num_tokens": 5907083.0, "reward": 0.9947597980499268, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9947597980499268, "reward_meter_std": 0.010325920768082142, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010325928218662739, "reward_total_composite_mean": 0.9947597980499268, "reward_total_composite_std": 0.010325920768082142, "reward_total_mean": 0.9947597980499268, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9947597980499268, "rewards/meter/std": 0.010325920768082142, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9947597980499268, "rewards/total_composite/std": 0.010325920768082142, "sampling/importance_sampling_ratio/max": 1.8022807836532593, "sampling/importance_sampling_ratio/mean": 1.0031919479370117, "sampling/importance_sampling_ratio/min": 0.26653534173965454, "sampling/sampling_logp_difference/max": 1.3222484588623047, "sampling/sampling_logp_difference/mean": 0.03754591941833496, "step": 2608 }, { "clip_ratio/high_max": 0.029262988013215363, "clip_ratio/high_mean": 0.029262988013215363, "clip_ratio/low_mean": 0.008980331476777792, "clip_ratio/low_min": 0.008980331476777792, "clip_ratio/region_mean": 0.038243319489993155, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.75, "completions/mean_terminated_length": 68.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.3395673893392086, "epoch": 0.10479174197694502, "frac_reward_zero_std": 0.0, "grad_norm": 5.4968791007995605, "learning_rate": 2.096969696969697e-06, "loss": 0.0134, "num_tokens": 5908961.0, "reward": 0.9976909756660461, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976909756660461, "reward_meter_std": 0.003610314568504691, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00361032597720623, "reward_total_composite_mean": 0.9976909756660461, "reward_total_composite_std": 0.003610314568504691, "reward_total_mean": 0.9976909756660461, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976909756660461, "rewards/meter/std": 0.003610314568504691, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976909756660461, "rewards/total_composite/std": 0.003610314568504691, "sampling/importance_sampling_ratio/max": 1.6728876829147339, "sampling/importance_sampling_ratio/mean": 1.0021377801895142, "sampling/importance_sampling_ratio/min": 0.19148778915405273, "sampling/sampling_logp_difference/max": 1.6529312133789062, "sampling/sampling_logp_difference/mean": 0.03931521996855736, "step": 2609 }, { "clip_ratio/high_max": 0.013105392456054688, "clip_ratio/high_mean": 0.013105392456054688, "clip_ratio/low_mean": 0.021181484800763428, "clip_ratio/low_min": 0.021181484800763428, "clip_ratio/region_mean": 0.034286877256818116, "completions/clipped_ratio": 0.0, "completions/max_length": 205.0, "completions/max_terminated_length": 205.0, "completions/mean_length": 191.0, "completions/mean_terminated_length": 191.0, "completions/min_length": 170.0, "completions/min_terminated_length": 170.0, "entropy": 0.49758507683873177, "epoch": 0.10483190745872997, "frac_reward_zero_std": 0.0, "grad_norm": 2.755105495452881, "learning_rate": 2.093939393939394e-06, "loss": -0.01, "num_tokens": 5911873.0, "reward": 0.832134485244751, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.08625820279121399, "reward_meter_mean": 0.9434575438499451, "reward_meter_std": 0.07025814801454544, "reward_repeat_penalty_mean": 0.9431818127632141, "reward_repeat_penalty_std": 0.08328413218259811, "reward_std": 0.10688581317663193, "reward_total_composite_mean": 0.832134485244751, "reward_total_composite_std": 0.10688581317663193, "reward_total_mean": 0.832134485244751, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.08625820279121399, "rewards/meter/mean": 0.9434575438499451, "rewards/meter/std": 0.07025814801454544, "rewards/repeat_penalty/mean": 0.9431818127632141, "rewards/repeat_penalty/std": 0.08328413218259811, "rewards/total_composite/mean": 0.832134485244751, "rewards/total_composite/std": 0.10688581317663193, "sampling/importance_sampling_ratio/max": 1.9386733770370483, "sampling/importance_sampling_ratio/mean": 1.007143259048462, "sampling/importance_sampling_ratio/min": 0.11968369781970978, "sampling/sampling_logp_difference/max": 2.1229028701782227, "sampling/sampling_logp_difference/mean": 0.057438045740127563, "step": 2610 }, { "clip_ratio/high_max": 0.00305292836856097, "clip_ratio/high_mean": 0.00305292836856097, "clip_ratio/low_mean": 0.007064696401357651, "clip_ratio/low_min": 0.007064696401357651, "clip_ratio/region_mean": 0.01011762476991862, "completions/clipped_ratio": 0.0, "completions/max_length": 249.0, "completions/max_terminated_length": 249.0, "completions/mean_length": 247.25, "completions/mean_terminated_length": 247.25, "completions/min_length": 245.0, "completions/min_terminated_length": 245.0, "entropy": 0.13410179316997528, "epoch": 0.10487207294051493, "frac_reward_zero_std": 0.0, "grad_norm": 1.116234540939331, "learning_rate": 2.090909090909091e-06, "loss": 0.005, "num_tokens": 5915555.0, "reward": 0.7778571844100952, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987331032752991, "reward_meter_std": 0.00010533389286138117, "reward_repeat_penalty_mean": 0.7788461446762085, "reward_repeat_penalty_std": 0.04929768666625023, "reward_std": 0.04919853433966637, "reward_total_composite_mean": 0.7778571844100952, "reward_total_composite_std": 0.04919851943850517, "reward_total_mean": 0.7778571844100952, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987331032752991, "rewards/meter/std": 0.00010533389286138117, "rewards/repeat_penalty/mean": 0.7788461446762085, "rewards/repeat_penalty/std": 0.04929768666625023, "rewards/total_composite/mean": 0.7778571844100952, "rewards/total_composite/std": 0.04919851943850517, "sampling/importance_sampling_ratio/max": 1.6523418426513672, "sampling/importance_sampling_ratio/mean": 1.0057003498077393, "sampling/importance_sampling_ratio/min": 0.2605571746826172, "sampling/sampling_logp_difference/max": 1.344933032989502, "sampling/sampling_logp_difference/mean": 0.01725967787206173, "step": 2611 }, { "clip_ratio/high_max": 0.004204352619126439, "clip_ratio/high_mean": 0.004204352619126439, "clip_ratio/low_mean": 0.011121554300189018, "clip_ratio/low_min": 0.011121554300189018, "clip_ratio/region_mean": 0.015325906919315457, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 57.75, "completions/mean_terminated_length": 57.75, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.08830610243603587, "epoch": 0.10491223842229988, "frac_reward_zero_std": 0.0, "grad_norm": 3.173931121826172, "learning_rate": 2.087878787878788e-06, "loss": -0.02, "num_tokens": 5917441.0, "reward": 0.995871365070343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.995871365070343, "reward_meter_std": 0.0009862068109214306, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009862068109214306, "reward_total_composite_mean": 0.995871365070343, "reward_total_composite_std": 0.0009862068109214306, "reward_total_mean": 0.995871365070343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.995871365070343, "rewards/meter/std": 0.0009862068109214306, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.995871365070343, "rewards/total_composite/std": 0.0009862068109214306, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0038785934448242, "sampling/importance_sampling_ratio/min": 0.4063306748867035, "sampling/sampling_logp_difference/max": 0.9005880355834961, "sampling/sampling_logp_difference/mean": 0.01807764358818531, "step": 2612 }, { "clip_ratio/high_max": 0.021714599570259452, "clip_ratio/high_mean": 0.021714599570259452, "clip_ratio/low_mean": 0.01867919461801648, "clip_ratio/low_min": 0.01867919461801648, "clip_ratio/region_mean": 0.04039379418827593, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 343.75, "completions/mean_terminated_length": 343.75, "completions/min_length": 326.0, "completions/min_terminated_length": 326.0, "entropy": 0.5114922970533371, "epoch": 0.10495240390408483, "frac_reward_zero_std": 0.0, "grad_norm": 2.3518784046173096, "learning_rate": 2.0848484848484852e-06, "loss": 0.0013, "num_tokens": 5922023.0, "reward": 0.6937090158462524, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.759615421295166, "reward_count_adherence_std": 0.027196412906050682, "reward_meter_mean": 0.990763247013092, "reward_meter_std": 0.004047461785376072, "reward_repeat_penalty_mean": 0.9213085174560547, "reward_repeat_penalty_std": 0.04907441511750221, "reward_std": 0.04994901269674301, "reward_total_composite_mean": 0.6937090158462524, "reward_total_composite_std": 0.049949031323194504, "reward_total_mean": 0.6937090158462524, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.759615421295166, "rewards/count_adherence/std": 0.027196412906050682, "rewards/meter/mean": 0.990763247013092, "rewards/meter/std": 0.004047461785376072, "rewards/repeat_penalty/mean": 0.9213085174560547, "rewards/repeat_penalty/std": 0.04907441511750221, "rewards/total_composite/mean": 0.6937090158462524, "rewards/total_composite/std": 0.049949031323194504, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0127009153366089, "sampling/importance_sampling_ratio/min": 0.21443161368370056, "sampling/sampling_logp_difference/max": 1.539764404296875, "sampling/sampling_logp_difference/mean": 0.05348372459411621, "step": 2613 }, { "clip_ratio/high_max": 0.02407875331118703, "clip_ratio/high_mean": 0.02407875331118703, "clip_ratio/low_mean": 0.009516695979982615, "clip_ratio/low_min": 0.009516695979982615, "clip_ratio/region_mean": 0.03359544929116964, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 118.75, "completions/mean_terminated_length": 118.75, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.20165002904832363, "epoch": 0.10499256938586979, "frac_reward_zero_std": 0.0, "grad_norm": 4.099353790283203, "learning_rate": 2.081818181818182e-06, "loss": 0.0036, "num_tokens": 5924333.0, "reward": 0.953779935836792, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9893277883529663, "reward_meter_std": 0.016893420368433, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06425920873880386, "reward_total_composite_mean": 0.953779935836792, "reward_total_composite_std": 0.06425921618938446, "reward_total_mean": 0.953779935836792, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9893277883529663, "rewards/meter/std": 0.016893420368433, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.953779935836792, "rewards/total_composite/std": 0.06425921618938446, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0034631490707397, "sampling/importance_sampling_ratio/min": 0.2708137333393097, "sampling/sampling_logp_difference/max": 1.3063240051269531, "sampling/sampling_logp_difference/mean": 0.028960946947336197, "step": 2614 }, { "clip_ratio/high_max": 0.015257316990755498, "clip_ratio/high_mean": 0.015257316990755498, "clip_ratio/low_mean": 0.010907472460530698, "clip_ratio/low_min": 0.010907472460530698, "clip_ratio/region_mean": 0.026164789451286197, "completions/clipped_ratio": 0.0, "completions/max_length": 77.0, "completions/max_terminated_length": 77.0, "completions/mean_length": 69.625, "completions/mean_terminated_length": 69.625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.3094298355281353, "epoch": 0.10503273486765474, "frac_reward_zero_std": 0.0, "grad_norm": 2.5716392993927, "learning_rate": 2.078787878787879e-06, "loss": -0.0063, "num_tokens": 5926114.0, "reward": 0.9630937576293945, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9630937576293945, "reward_meter_std": 0.028052091598510742, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.028052086010575294, "reward_total_composite_mean": 0.9630937576293945, "reward_total_composite_std": 0.028052091598510742, "reward_total_mean": 0.9630937576293945, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9630937576293945, "rewards/meter/std": 0.028052091598510742, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9630937576293945, "rewards/total_composite/std": 0.028052091598510742, "sampling/importance_sampling_ratio/max": 1.9397474527359009, "sampling/importance_sampling_ratio/mean": 1.0128580331802368, "sampling/importance_sampling_ratio/min": 0.16488294303417206, "sampling/sampling_logp_difference/max": 1.8025195598602295, "sampling/sampling_logp_difference/mean": 0.04609367623925209, "step": 2615 }, { "clip_ratio/high_max": 0.003906250058207661, "clip_ratio/high_mean": 0.003906250058207661, "clip_ratio/low_mean": 0.0007812500116415322, "clip_ratio/low_min": 0.0007812500116415322, "clip_ratio/region_mean": 0.004687500069849193, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 159.5, "completions/mean_terminated_length": 159.5, "completions/min_length": 156.0, "completions/min_terminated_length": 156.0, "entropy": 0.09610292688012123, "epoch": 0.1050729003494397, "frac_reward_zero_std": 0.0, "grad_norm": 1.3603968620300293, "learning_rate": 2.075757575757576e-06, "loss": -0.0043, "num_tokens": 5928878.0, "reward": 0.8684781789779663, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9918752908706665, "reward_meter_std": 0.01707914099097252, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.07120776921510696, "reward_std": 0.07916972041130066, "reward_total_composite_mean": 0.8684781789779663, "reward_total_composite_std": 0.07916970551013947, "reward_total_mean": 0.8684781789779663, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9918752908706665, "rewards/meter/std": 0.01707914099097252, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.07120776921510696, "rewards/total_composite/mean": 0.8684781789779663, "rewards/total_composite/std": 0.07916970551013947, "sampling/importance_sampling_ratio/max": 1.4536888599395752, "sampling/importance_sampling_ratio/mean": 1.0036389827728271, "sampling/importance_sampling_ratio/min": 0.582060694694519, "sampling/sampling_logp_difference/max": 0.5411806106567383, "sampling/sampling_logp_difference/mean": 0.00969278160482645, "step": 2616 }, { "clip_ratio/high_max": 0.00300038093701005, "clip_ratio/high_mean": 0.00300038093701005, "clip_ratio/low_mean": 0.002922236453741789, "clip_ratio/low_min": 0.002922236453741789, "clip_ratio/region_mean": 0.005922617390751839, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 127.25, "completions/mean_terminated_length": 127.25, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.11067105457186699, "epoch": 0.10511306583122465, "frac_reward_zero_std": 0.0, "grad_norm": 1.5142834186553955, "learning_rate": 2.072727272727273e-06, "loss": -0.0021, "num_tokens": 5931264.0, "reward": 0.9442130923271179, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976711869239807, "reward_meter_std": 0.0005347781116142869, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07360443472862244, "reward_total_composite_mean": 0.9442130923271179, "reward_total_composite_std": 0.07360443472862244, "reward_total_mean": 0.9442130923271179, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976711869239807, "rewards/meter/std": 0.0005347781116142869, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9442130923271179, "rewards/total_composite/std": 0.07360443472862244, "sampling/importance_sampling_ratio/max": 1.5947275161743164, "sampling/importance_sampling_ratio/mean": 1.001734733581543, "sampling/importance_sampling_ratio/min": 0.4548099637031555, "sampling/sampling_logp_difference/max": 0.7878756523132324, "sampling/sampling_logp_difference/mean": 0.011169607751071453, "step": 2617 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.002262885362142697, "epoch": 0.1051532313130096, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.06969696969697e-06, "loss": 0.0, "num_tokens": 5932856.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.002797245979309, "sampling/importance_sampling_ratio/mean": 1.000277042388916, "sampling/importance_sampling_ratio/min": 1.0, "sampling/sampling_logp_difference/max": 0.002793335122987628, "sampling/sampling_logp_difference/mean": 0.00027682288782671094, "step": 2618 }, { "clip_ratio/high_max": 0.030287093366496265, "clip_ratio/high_mean": 0.030287093366496265, "clip_ratio/low_mean": 0.011174242943525314, "clip_ratio/low_min": 0.011174242943525314, "clip_ratio/region_mean": 0.04146133631002158, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 90.75, "completions/mean_terminated_length": 90.75, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.22051831148564816, "epoch": 0.10519339679479456, "frac_reward_zero_std": 0.0, "grad_norm": 3.8841686248779297, "learning_rate": 2.0666666666666666e-06, "loss": -0.01, "num_tokens": 5935062.0, "reward": 0.9945749640464783, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9945749640464783, "reward_meter_std": 0.00539820222184062, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005398200359195471, "reward_total_composite_mean": 0.9945749640464783, "reward_total_composite_std": 0.00539820222184062, "reward_total_mean": 0.9945749640464783, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9945749640464783, "rewards/meter/std": 0.00539820222184062, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9945749640464783, "rewards/total_composite/std": 0.00539820222184062, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023733377456665, "sampling/importance_sampling_ratio/min": 0.06795818358659744, "sampling/sampling_logp_difference/max": 2.6888628005981445, "sampling/sampling_logp_difference/mean": 0.038317468017339706, "step": 2619 }, { "clip_ratio/high_max": 0.016332621686160564, "clip_ratio/high_mean": 0.016332621686160564, "clip_ratio/low_mean": 0.009193609352223575, "clip_ratio/low_min": 0.009193609352223575, "clip_ratio/region_mean": 0.02552623103838414, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 298.625, "completions/mean_terminated_length": 298.625, "completions/min_length": 293.0, "completions/min_terminated_length": 293.0, "entropy": 0.2537948340177536, "epoch": 0.10523356227657951, "frac_reward_zero_std": 0.0, "grad_norm": 1.9527921676635742, "learning_rate": 2.063636363636364e-06, "loss": 0.0003, "num_tokens": 5939163.0, "reward": 0.9610700011253357, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977849721908569, "reward_meter_std": 0.0011817796621471643, "reward_repeat_penalty_mean": 0.9632352590560913, "reward_repeat_penalty_std": 0.04376610368490219, "reward_std": 0.042887069284915924, "reward_total_composite_mean": 0.9610700011253357, "reward_total_composite_std": 0.042887065559625626, "reward_total_mean": 0.9610700011253357, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977849721908569, "rewards/meter/std": 0.0011817796621471643, "rewards/repeat_penalty/mean": 0.9632352590560913, "rewards/repeat_penalty/std": 0.04376610368490219, "rewards/total_composite/mean": 0.9610700011253357, "rewards/total_composite/std": 0.042887065559625626, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0045593976974487, "sampling/importance_sampling_ratio/min": 0.09144318848848343, "sampling/sampling_logp_difference/max": 2.3920373916625977, "sampling/sampling_logp_difference/mean": 0.0335632786154747, "step": 2620 }, { "clip_ratio/high_max": 0.004175611422397196, "clip_ratio/high_mean": 0.004175611422397196, "clip_ratio/low_mean": 0.006353942735586315, "clip_ratio/low_min": 0.006353942735586315, "clip_ratio/region_mean": 0.010529554157983512, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 349.125, "completions/mean_terminated_length": 349.125, "completions/min_length": 319.0, "completions/min_terminated_length": 319.0, "entropy": 0.18446156568825245, "epoch": 0.10527372775836447, "frac_reward_zero_std": 0.0, "grad_norm": 1.2172192335128784, "learning_rate": 2.0606060606060607e-06, "loss": 0.0049, "num_tokens": 5943660.0, "reward": 0.7889921069145203, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.987500011920929, "reward_count_adherence_std": 0.0353553481400013, "reward_meter_mean": 0.9986586570739746, "reward_meter_std": 0.0001423863577656448, "reward_repeat_penalty_mean": 0.8010836243629456, "reward_repeat_penalty_std": 0.058759476989507675, "reward_std": 0.04864564538002014, "reward_total_composite_mean": 0.7889921069145203, "reward_total_composite_std": 0.04864564165472984, "reward_total_mean": 0.7889921069145203, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.987500011920929, "rewards/count_adherence/std": 0.0353553481400013, "rewards/meter/mean": 0.9986586570739746, "rewards/meter/std": 0.0001423863577656448, "rewards/repeat_penalty/mean": 0.8010836243629456, "rewards/repeat_penalty/std": 0.058759476989507675, "rewards/total_composite/mean": 0.7889921069145203, "rewards/total_composite/std": 0.04864564165472984, "sampling/importance_sampling_ratio/max": 1.901794672012329, "sampling/importance_sampling_ratio/mean": 1.0064207315444946, "sampling/importance_sampling_ratio/min": 0.24801570177078247, "sampling/sampling_logp_difference/max": 1.3942632675170898, "sampling/sampling_logp_difference/mean": 0.02395707368850708, "step": 2621 }, { "clip_ratio/high_max": 0.016635986510664225, "clip_ratio/high_mean": 0.016635986510664225, "clip_ratio/low_mean": 0.006287686061114073, "clip_ratio/low_min": 0.006287686061114073, "clip_ratio/region_mean": 0.022923672571778297, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 59.875, "completions/mean_terminated_length": 59.875, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.1663585538044572, "epoch": 0.10531389324014942, "frac_reward_zero_std": 0.0, "grad_norm": 5.598334789276123, "learning_rate": 2.0575757575757576e-06, "loss": -0.0037, "num_tokens": 5945475.0, "reward": 0.9960442185401917, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9960442185401917, "reward_meter_std": 0.0010061725042760372, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010061829816550016, "reward_total_composite_mean": 0.9960442185401917, "reward_total_composite_std": 0.0010061725042760372, "reward_total_mean": 0.9960442185401917, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9960442185401917, "rewards/meter/std": 0.0010061725042760372, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9960442185401917, "rewards/total_composite/std": 0.0010061725042760372, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9990103244781494, "sampling/importance_sampling_ratio/min": 0.20777083933353424, "sampling/sampling_logp_difference/max": 1.571319580078125, "sampling/sampling_logp_difference/mean": 0.036090608686208725, "step": 2622 }, { "clip_ratio/high_max": 0.0052327855955809355, "clip_ratio/high_mean": 0.0052327855955809355, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/region_mean": 0.006993348943069577, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 71.75, "completions/mean_terminated_length": 71.75, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.12103518098592758, "epoch": 0.10535405872193437, "frac_reward_zero_std": 0.0, "grad_norm": 3.970113515853882, "learning_rate": 2.054545454545455e-06, "loss": 0.003, "num_tokens": 5947273.0, "reward": 0.9978034496307373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978034496307373, "reward_meter_std": 0.004221748560667038, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004221739247441292, "reward_total_composite_mean": 0.9978034496307373, "reward_total_composite_std": 0.004221748560667038, "reward_total_mean": 0.9978034496307373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978034496307373, "rewards/meter/std": 0.004221748560667038, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978034496307373, "rewards/total_composite/std": 0.004221748560667038, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0029164552688599, "sampling/importance_sampling_ratio/min": 0.39022698998451233, "sampling/sampling_logp_difference/max": 0.9410266876220703, "sampling/sampling_logp_difference/mean": 0.020419280976057053, "step": 2623 }, { "clip_ratio/high_max": 0.011576013872399926, "clip_ratio/high_mean": 0.011576013872399926, "clip_ratio/low_mean": 0.041920434683561325, "clip_ratio/low_min": 0.041920434683561325, "clip_ratio/region_mean": 0.05349644855596125, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 64.75, "completions/mean_terminated_length": 64.75, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.3771820291876793, "epoch": 0.10539422420371933, "frac_reward_zero_std": 0.0, "grad_norm": 9.467141151428223, "learning_rate": 2.0515151515151517e-06, "loss": 0.0286, "num_tokens": 5948959.0, "reward": 0.017361678183078766, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.017361678183078766, "reward_meter_std": 0.014113202691078186, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014113202691078186, "reward_total_composite_mean": 0.017361678183078766, "reward_total_composite_std": 0.014113202691078186, "reward_total_mean": 0.017361678183078766, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.017361678183078766, "rewards/meter/std": 0.014113202691078186, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.017361678183078766, "rewards/total_composite/std": 0.014113202691078186, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9971011877059937, "sampling/importance_sampling_ratio/min": 0.12875375151634216, "sampling/sampling_logp_difference/max": 2.049853563308716, "sampling/sampling_logp_difference/mean": 0.08726956695318222, "step": 2624 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 37.0, "completions/mean_terminated_length": 37.0, "completions/min_length": 37.0, "completions/min_terminated_length": 37.0, "entropy": 0.022779989056289196, "epoch": 0.10543438968550428, "frac_reward_zero_std": 0.0, "grad_norm": 1.3292933702468872, "learning_rate": 2.0484848484848485e-06, "loss": -0.0006, "num_tokens": 5950439.0, "reward": 0.9995855093002319, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9995855093002319, "reward_meter_std": 5.365296237869188e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.3658961405744776e-05, "reward_total_composite_mean": 0.9995855093002319, "reward_total_composite_std": 5.365296237869188e-05, "reward_total_mean": 0.9995855093002319, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9995855093002319, "rewards/meter/std": 5.365296237869188e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995855093002319, "rewards/total_composite/std": 5.365296237869188e-05, "sampling/importance_sampling_ratio/max": 1.0283234119415283, "sampling/importance_sampling_ratio/mean": 1.000502109527588, "sampling/importance_sampling_ratio/min": 0.5255436897277832, "sampling/sampling_logp_difference/max": 0.6433219909667969, "sampling/sampling_logp_difference/mean": 0.004437229596078396, "step": 2625 }, { "clip_ratio/high_max": 0.009794845129363239, "clip_ratio/high_mean": 0.009794845129363239, "clip_ratio/low_mean": 0.008605769136920571, "clip_ratio/low_min": 0.008605769136920571, "clip_ratio/region_mean": 0.01840061426628381, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 102.375, "completions/mean_terminated_length": 102.375, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.3144999463111162, "epoch": 0.10547455516728924, "frac_reward_zero_std": 0.0, "grad_norm": 2.536959409713745, "learning_rate": 2.0454545454545457e-06, "loss": 0.0077, "num_tokens": 5952538.0, "reward": 0.9648023843765259, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9648023843765259, "reward_meter_std": 0.04039134085178375, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04039134085178375, "reward_total_composite_mean": 0.9648023843765259, "reward_total_composite_std": 0.04039134085178375, "reward_total_mean": 0.9648023843765259, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9648023843765259, "rewards/meter/std": 0.04039134085178375, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9648023843765259, "rewards/total_composite/std": 0.04039134085178375, "sampling/importance_sampling_ratio/max": 1.6411556005477905, "sampling/importance_sampling_ratio/mean": 1.0097938776016235, "sampling/importance_sampling_ratio/min": 0.3631163537502289, "sampling/sampling_logp_difference/max": 1.0130319595336914, "sampling/sampling_logp_difference/mean": 0.03456467390060425, "step": 2626 }, { "clip_ratio/high_max": 0.012847222620621324, "clip_ratio/high_mean": 0.012847222620621324, "clip_ratio/low_mean": 0.012954836362041533, "clip_ratio/low_min": 0.012954836362041533, "clip_ratio/region_mean": 0.025802058982662857, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 60.5, "completions/mean_terminated_length": 60.5, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.2536655478179455, "epoch": 0.10551472064907419, "frac_reward_zero_std": 0.0, "grad_norm": 6.214305877685547, "learning_rate": 2.0424242424242426e-06, "loss": 0.2115, "num_tokens": 5954150.0, "reward": 0.24602118134498596, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.25, "reward_count_adherence_std": 0.4629100561141968, "reward_meter_mean": 0.9953294396400452, "reward_meter_std": 0.010616513900458813, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4556134045124054, "reward_total_composite_mean": 0.24602118134498596, "reward_total_composite_std": 0.4556134343147278, "reward_total_mean": 0.24602118134498596, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.25, "rewards/count_adherence/std": 0.4629100561141968, "rewards/meter/mean": 0.9953294396400452, "rewards/meter/std": 0.010616513900458813, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.24602118134498596, "rewards/total_composite/std": 0.4556134343147278, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0146692991256714, "sampling/importance_sampling_ratio/min": 0.16502384841442108, "sampling/sampling_logp_difference/max": 1.8016653060913086, "sampling/sampling_logp_difference/mean": 0.03473228961229324, "step": 2627 }, { "clip_ratio/high_max": 0.0423777480609715, "clip_ratio/high_mean": 0.0423777480609715, "clip_ratio/low_mean": 0.00927626434713602, "clip_ratio/low_min": 0.00927626434713602, "clip_ratio/region_mean": 0.05165401240810752, "completions/clipped_ratio": 0.0, "completions/max_length": 243.0, "completions/max_terminated_length": 243.0, "completions/mean_length": 239.75, "completions/mean_terminated_length": 239.75, "completions/min_length": 233.0, "completions/min_terminated_length": 233.0, "entropy": 0.7806448265910149, "epoch": 0.10555488613085914, "frac_reward_zero_std": 0.0, "grad_norm": 3.508739709854126, "learning_rate": 2.03939393939394e-06, "loss": 0.014, "num_tokens": 5957532.0, "reward": 0.9953933954238892, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9953933954238892, "reward_meter_std": 0.00770949712023139, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007709511090070009, "reward_total_composite_mean": 0.9953933954238892, "reward_total_composite_std": 0.00770949712023139, "reward_total_mean": 0.9953933954238892, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9953933954238892, "rewards/meter/std": 0.00770949712023139, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9953933954238892, "rewards/total_composite/std": 0.00770949712023139, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0158778429031372, "sampling/importance_sampling_ratio/min": 0.04539341852068901, "sampling/sampling_logp_difference/max": 3.092388153076172, "sampling/sampling_logp_difference/mean": 0.07987385243177414, "step": 2628 }, { "clip_ratio/high_max": 0.018142351880669594, "clip_ratio/high_mean": 0.018142351880669594, "clip_ratio/low_mean": 0.012730259681120515, "clip_ratio/low_min": 0.012730259681120515, "clip_ratio/region_mean": 0.03087261156179011, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 317.875, "completions/mean_terminated_length": 317.875, "completions/min_length": 300.0, "completions/min_terminated_length": 300.0, "entropy": 0.4582854099571705, "epoch": 0.1055950516126441, "frac_reward_zero_std": 0.0, "grad_norm": 1.9617708921432495, "learning_rate": 2.0363636363636367e-06, "loss": 0.014, "num_tokens": 5961811.0, "reward": 0.927051305770874, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.9979603886604309, "reward_meter_std": 0.0011684981873258948, "reward_repeat_penalty_mean": 0.9593137502670288, "reward_repeat_penalty_std": 0.06067804619669914, "reward_std": 0.07678526639938354, "reward_total_composite_mean": 0.927051305770874, "reward_total_composite_std": 0.07678526639938354, "reward_total_mean": 0.927051305770874, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.9979603886604309, "rewards/meter/std": 0.0011684981873258948, "rewards/repeat_penalty/mean": 0.9593137502670288, "rewards/repeat_penalty/std": 0.06067804619669914, "rewards/total_composite/mean": 0.927051305770874, "rewards/total_composite/std": 0.07678526639938354, "sampling/importance_sampling_ratio/max": 1.913344383239746, "sampling/importance_sampling_ratio/mean": 1.0117024183273315, "sampling/importance_sampling_ratio/min": 0.19862042367458344, "sampling/sampling_logp_difference/max": 1.6163597106933594, "sampling/sampling_logp_difference/mean": 0.054092179983854294, "step": 2629 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.015519737149588764, "epoch": 0.10563521709442905, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.0333333333333335e-06, "loss": 0.0, "num_tokens": 5963587.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.1619328260421753, "sampling/importance_sampling_ratio/mean": 0.9990300536155701, "sampling/importance_sampling_ratio/min": 0.6878003478050232, "sampling/sampling_logp_difference/max": 0.3742567002773285, "sampling/sampling_logp_difference/mean": 0.0030186243820935488, "step": 2630 }, { "clip_ratio/high_max": 0.006118966557551175, "clip_ratio/high_mean": 0.006118966557551175, "clip_ratio/low_mean": 0.0017241379246115685, "clip_ratio/low_min": 0.0017241379246115685, "clip_ratio/region_mean": 0.007843104482162744, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 143.25, "completions/mean_terminated_length": 143.25, "completions/min_length": 142.0, "completions/min_terminated_length": 142.0, "entropy": 0.09556546807289124, "epoch": 0.105675382576214, "frac_reward_zero_std": 0.0, "grad_norm": 1.6037280559539795, "learning_rate": 2.0303030303030303e-06, "loss": 0.0062, "num_tokens": 5966253.0, "reward": 0.838421106338501, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989732503890991, "reward_meter_std": 0.00014036186621524394, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.050409551709890366, "reward_total_composite_mean": 0.838421106338501, "reward_total_composite_std": 0.05040955916047096, "reward_total_mean": 0.838421106338501, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989732503890991, "rewards/meter/std": 0.00014036186621524394, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.838421106338501, "rewards/total_composite/std": 0.05040955916047096, "sampling/importance_sampling_ratio/max": 1.676928997039795, "sampling/importance_sampling_ratio/mean": 1.008257508277893, "sampling/importance_sampling_ratio/min": 0.36321523785591125, "sampling/sampling_logp_difference/max": 1.0127596855163574, "sampling/sampling_logp_difference/mean": 0.013544322922825813, "step": 2631 }, { "clip_ratio/high_max": 0.015901052742265165, "clip_ratio/high_mean": 0.015901052742265165, "clip_ratio/low_mean": 0.023193509317934513, "clip_ratio/low_min": 0.023193509317934513, "clip_ratio/region_mean": 0.03909456206019968, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 70.25, "completions/mean_terminated_length": 70.25, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.3492947220802307, "epoch": 0.10571554805799896, "frac_reward_zero_std": 0.0, "grad_norm": 5.14970064163208, "learning_rate": 2.0272727272727276e-06, "loss": -0.007, "num_tokens": 5968023.0, "reward": 0.9124512672424316, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9124512672424316, "reward_meter_std": 0.0726349726319313, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0726349726319313, "reward_total_composite_mean": 0.9124512672424316, "reward_total_composite_std": 0.0726349726319313, "reward_total_mean": 0.9124512672424316, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9124512672424316, "rewards/meter/std": 0.0726349726319313, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9124512672424316, "rewards/total_composite/std": 0.0726349726319313, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0102332830429077, "sampling/importance_sampling_ratio/min": 0.14260892570018768, "sampling/sampling_logp_difference/max": 1.9476492404937744, "sampling/sampling_logp_difference/mean": 0.04469385743141174, "step": 2632 }, { "clip_ratio/high_max": 0.011128432815894485, "clip_ratio/high_mean": 0.011128432815894485, "clip_ratio/low_mean": 0.010007655015215278, "clip_ratio/low_min": 0.010007655015215278, "clip_ratio/region_mean": 0.021136087831109762, "completions/clipped_ratio": 0.0, "completions/max_length": 141.0, "completions/max_terminated_length": 141.0, "completions/mean_length": 136.25, "completions/mean_terminated_length": 136.25, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.3410081285983324, "epoch": 0.10575571353978391, "frac_reward_zero_std": 0.0, "grad_norm": 2.2837815284729004, "learning_rate": 2.0242424242424244e-06, "loss": 0.0122, "num_tokens": 5970465.0, "reward": 0.8844000101089478, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9715265035629272, "reward_meter_std": 0.011229481548070908, "reward_repeat_penalty_mean": 0.910714328289032, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.06688975542783737, "reward_total_composite_mean": 0.8844000101089478, "reward_total_composite_std": 0.06688975542783737, "reward_total_mean": 0.8844000101089478, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9715265035629272, "rewards/meter/std": 0.011229481548070908, "rewards/repeat_penalty/mean": 0.910714328289032, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.8844000101089478, "rewards/total_composite/std": 0.06688975542783737, "sampling/importance_sampling_ratio/max": 1.7595460414886475, "sampling/importance_sampling_ratio/mean": 1.0096205472946167, "sampling/importance_sampling_ratio/min": 0.16036662459373474, "sampling/sampling_logp_difference/max": 1.8302927017211914, "sampling/sampling_logp_difference/mean": 0.03914301469922066, "step": 2633 }, { "clip_ratio/high_max": 0.04050721391104162, "clip_ratio/high_mean": 0.04050721391104162, "clip_ratio/low_mean": 0.0086805559694767, "clip_ratio/low_min": 0.0086805559694767, "clip_ratio/region_mean": 0.04918776988051832, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.0, "completions/mean_terminated_length": 71.0, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.5871141031384468, "epoch": 0.10579587902156887, "frac_reward_zero_std": 0.0, "grad_norm": 3.726565361022949, "learning_rate": 2.0212121212121212e-06, "loss": 0.0021, "num_tokens": 5972297.0, "reward": 0.9706488251686096, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9706488251686096, "reward_meter_std": 0.049138810485601425, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.049138810485601425, "reward_total_composite_mean": 0.9706488251686096, "reward_total_composite_std": 0.049138810485601425, "reward_total_mean": 0.9706488251686096, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9706488251686096, "rewards/meter/std": 0.049138810485601425, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9706488251686096, "rewards/total_composite/std": 0.049138810485601425, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0056947469711304, "sampling/importance_sampling_ratio/min": 0.12379039824008942, "sampling/sampling_logp_difference/max": 2.089165449142456, "sampling/sampling_logp_difference/mean": 0.07419008761644363, "step": 2634 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.014239851152524352, "epoch": 0.10583604450335382, "frac_reward_zero_std": 0.0, "grad_norm": 0.053679000586271286, "learning_rate": 2.0181818181818185e-06, "loss": 0.0001, "num_tokens": 5974177.0, "reward": 0.9993990659713745, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993990659713745, "reward_meter_std": 1.0326285746486974e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.023619461193448e-06, "reward_total_composite_mean": 0.9993990659713745, "reward_total_composite_std": 1.0326285746486974e-06, "reward_total_mean": 0.9993990659713745, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993990659713745, "rewards/meter/std": 1.0326285746486974e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993990659713745, "rewards/total_composite/std": 1.0326285746486974e-06, "sampling/importance_sampling_ratio/max": 1.6687591075897217, "sampling/importance_sampling_ratio/mean": 1.0016577243804932, "sampling/importance_sampling_ratio/min": 0.9006934762001038, "sampling/sampling_logp_difference/max": 0.5120803117752075, "sampling/sampling_logp_difference/mean": 0.002415498485788703, "step": 2635 }, { "clip_ratio/high_max": 0.008138859295286238, "clip_ratio/high_mean": 0.008138859295286238, "clip_ratio/low_mean": 0.009945004130713642, "clip_ratio/low_min": 0.009945004130713642, "clip_ratio/region_mean": 0.01808386342599988, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 90.375, "completions/mean_terminated_length": 90.375, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.12905630934983492, "epoch": 0.10587620998513878, "frac_reward_zero_std": 0.0, "grad_norm": 2.7606544494628906, "learning_rate": 2.0151515151515153e-06, "loss": -0.0168, "num_tokens": 5976300.0, "reward": 0.9972773790359497, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972773790359497, "reward_meter_std": 0.0004967614077031612, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004967702552676201, "reward_total_composite_mean": 0.9972773790359497, "reward_total_composite_std": 0.0004967614077031612, "reward_total_mean": 0.9972773790359497, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972773790359497, "rewards/meter/std": 0.0004967614077031612, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972773790359497, "rewards/total_composite/std": 0.0004967614077031612, "sampling/importance_sampling_ratio/max": 1.9276334047317505, "sampling/importance_sampling_ratio/mean": 1.0045125484466553, "sampling/importance_sampling_ratio/min": 0.20726369321346283, "sampling/sampling_logp_difference/max": 1.573763370513916, "sampling/sampling_logp_difference/mean": 0.025844357907772064, "step": 2636 }, { "clip_ratio/high_max": 0.007327559753321111, "clip_ratio/high_mean": 0.007327559753321111, "clip_ratio/low_mean": 0.010715189506299794, "clip_ratio/low_min": 0.010715189506299794, "clip_ratio/region_mean": 0.018042749259620905, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 117.875, "completions/mean_terminated_length": 117.875, "completions/min_length": 113.0, "completions/min_terminated_length": 113.0, "entropy": 0.33568100817501545, "epoch": 0.10591637546692373, "frac_reward_zero_std": 0.0, "grad_norm": 2.6201107501983643, "learning_rate": 2.012121212121212e-06, "loss": -0.0064, "num_tokens": 5978451.0, "reward": 0.9981694221496582, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981694221496582, "reward_meter_std": 0.0008605656330473721, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008605782641097903, "reward_total_composite_mean": 0.9981694221496582, "reward_total_composite_std": 0.0008605656330473721, "reward_total_mean": 0.9981694221496582, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981694221496582, "rewards/meter/std": 0.0008605656330473721, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981694221496582, "rewards/total_composite/std": 0.0008605656330473721, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0066555738449097, "sampling/importance_sampling_ratio/min": 0.27018171548843384, "sampling/sampling_logp_difference/max": 1.3086605072021484, "sampling/sampling_logp_difference/mean": 0.044327329844236374, "step": 2637 }, { "clip_ratio/high_max": 0.017471515806391835, "clip_ratio/high_mean": 0.017471515806391835, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/region_mean": 0.02076098951511085, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 35.75, "completions/mean_terminated_length": 35.75, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.1990708950906992, "epoch": 0.10595654094870868, "frac_reward_zero_std": 0.0, "grad_norm": 6.669706344604492, "learning_rate": 2.009090909090909e-06, "loss": 0.0217, "num_tokens": 5979937.0, "reward": 0.9986928701400757, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986928701400757, "reward_meter_std": 0.002228210447356105, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002228200202807784, "reward_total_composite_mean": 0.9986928701400757, "reward_total_composite_std": 0.002228210447356105, "reward_total_mean": 0.9986928701400757, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986928701400757, "rewards/meter/std": 0.002228210447356105, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986928701400757, "rewards/total_composite/std": 0.002228210447356105, "sampling/importance_sampling_ratio/max": 1.4079704284667969, "sampling/importance_sampling_ratio/mean": 1.0115097761154175, "sampling/importance_sampling_ratio/min": 0.4690699875354767, "sampling/sampling_logp_difference/max": 0.7570033073425293, "sampling/sampling_logp_difference/mean": 0.02272917330265045, "step": 2638 }, { "clip_ratio/high_max": 0.004762336844578385, "clip_ratio/high_mean": 0.004762336844578385, "clip_ratio/low_mean": 0.003537735901772976, "clip_ratio/low_min": 0.003537735901772976, "clip_ratio/region_mean": 0.008300072746351361, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 106.75, "completions/mean_terminated_length": 106.75, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.09720816975459456, "epoch": 0.10599670643049364, "frac_reward_zero_std": 0.0, "grad_norm": 1.0811501741409302, "learning_rate": 2.0060606060606062e-06, "loss": -0.0005, "num_tokens": 5982303.0, "reward": 0.9990361332893372, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990361332893372, "reward_meter_std": 0.00013772756210528314, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00013771916565019637, "reward_total_composite_mean": 0.9990361332893372, "reward_total_composite_std": 0.00013772756210528314, "reward_total_mean": 0.9990361332893372, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990361332893372, "rewards/meter/std": 0.00013772756210528314, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990361332893372, "rewards/total_composite/std": 0.00013772756210528314, "sampling/importance_sampling_ratio/max": 1.4171427488327026, "sampling/importance_sampling_ratio/mean": 1.0037332773208618, "sampling/importance_sampling_ratio/min": 0.21328236162662506, "sampling/sampling_logp_difference/max": 1.5451383590698242, "sampling/sampling_logp_difference/mean": 0.014354676008224487, "step": 2639 }, { "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0017361111240461469, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.5, "completions/mean_terminated_length": 71.5, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.04928429890424013, "epoch": 0.10603687191227859, "frac_reward_zero_std": 0.0, "grad_norm": 1.591088891029358, "learning_rate": 2.0030303030303035e-06, "loss": 0.0003, "num_tokens": 5984115.0, "reward": 0.9991612434387207, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991612434387207, "reward_meter_std": 0.00017454942280892283, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00017454590124543756, "reward_total_composite_mean": 0.9991612434387207, "reward_total_composite_std": 0.00017454942280892283, "reward_total_mean": 0.9991612434387207, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991612434387207, "rewards/meter/std": 0.00017454942280892283, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991612434387207, "rewards/total_composite/std": 0.00017454942280892283, "sampling/importance_sampling_ratio/max": 1.096482753753662, "sampling/importance_sampling_ratio/mean": 1.001218557357788, "sampling/importance_sampling_ratio/min": 0.4362854063510895, "sampling/sampling_logp_difference/max": 0.8294587135314941, "sampling/sampling_logp_difference/mean": 0.008529691025614738, "step": 2640 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.00390625, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.01789803896099329, "epoch": 0.10607703739406354, "frac_reward_zero_std": 0.0, "grad_norm": 0.12622658908367157, "learning_rate": 2.0000000000000003e-06, "loss": 0.0002, "num_tokens": 5985907.0, "reward": 0.9993976354598999, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993976354598999, "reward_meter_std": 3.0407550184463616e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.030640073120594e-06, "reward_total_composite_mean": 0.9993976354598999, "reward_total_composite_std": 3.0407550184463616e-06, "reward_total_mean": 0.9993976354598999, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993976354598999, "rewards/meter/std": 3.0407550184463616e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993976354598999, "rewards/total_composite/std": 3.0407550184463616e-06, "sampling/importance_sampling_ratio/max": 1.6674405336380005, "sampling/importance_sampling_ratio/mean": 0.9996961951255798, "sampling/importance_sampling_ratio/min": 0.5132443308830261, "sampling/sampling_logp_difference/max": 0.6670032739639282, "sampling/sampling_logp_difference/mean": 0.004023827612400055, "step": 2641 }, { "clip_ratio/high_max": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/low_mean": 0.00390625, "clip_ratio/low_min": 0.00390625, "clip_ratio/region_mean": 0.005859375, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.01494286535307765, "epoch": 0.1061172028758485, "frac_reward_zero_std": 0.0, "grad_norm": 0.3109838664531708, "learning_rate": 1.996969696969697e-06, "loss": 0.0, "num_tokens": 5987731.0, "reward": 0.9993950724601746, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993950724601746, "reward_meter_std": 5.620834144792752e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.6233166105812415e-06, "reward_total_composite_mean": 0.9993950724601746, "reward_total_composite_std": 5.620834144792752e-06, "reward_total_mean": 0.9993950724601746, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993950724601746, "rewards/meter/std": 5.620834144792752e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993950724601746, "rewards/total_composite/std": 5.620834144792752e-06, "sampling/importance_sampling_ratio/max": 1.4890631437301636, "sampling/importance_sampling_ratio/mean": 0.9969460964202881, "sampling/importance_sampling_ratio/min": 0.43948376178741455, "sampling/sampling_logp_difference/max": 0.8221545219421387, "sampling/sampling_logp_difference/mean": 0.00695427879691124, "step": 2642 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0020185734319966286, "epoch": 0.10615736835763345, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.993939393939394e-06, "loss": 0.0, "num_tokens": 5989459.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.003127932548523, "sampling/importance_sampling_ratio/mean": 1.0002434253692627, "sampling/importance_sampling_ratio/min": 0.9998295307159424, "sampling/sampling_logp_difference/max": 0.003122988622635603, "sampling/sampling_logp_difference/mean": 0.00024456807295791805, "step": 2643 }, { "clip_ratio/high_max": 0.006403850042261183, "clip_ratio/high_mean": 0.006403850042261183, "clip_ratio/low_mean": 0.0012886597542092204, "clip_ratio/low_min": 0.0012886597542092204, "clip_ratio/region_mean": 0.007692509796470404, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.5, "completions/mean_terminated_length": 97.5, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.07244368549436331, "epoch": 0.1061975338394184, "frac_reward_zero_std": 0.0, "grad_norm": 0.6552127003669739, "learning_rate": 1.9909090909090913e-06, "loss": -0.0016, "num_tokens": 5991639.0, "reward": 0.9979605674743652, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979605674743652, "reward_meter_std": 0.00010473600559635088, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010474038572283462, "reward_total_composite_mean": 0.9979605674743652, "reward_total_composite_std": 0.00010473600559635088, "reward_total_mean": 0.9979605674743652, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979605674743652, "rewards/meter/std": 0.00010473600559635088, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979605674743652, "rewards/total_composite/std": 0.00010473600559635088, "sampling/importance_sampling_ratio/max": 1.362816333770752, "sampling/importance_sampling_ratio/mean": 0.9999778270721436, "sampling/importance_sampling_ratio/min": 0.31899797916412354, "sampling/sampling_logp_difference/max": 1.1425704956054688, "sampling/sampling_logp_difference/mean": 0.011263424530625343, "step": 2644 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018656715983524919, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.04345296695828438, "epoch": 0.10623769932120336, "frac_reward_zero_std": 0.0, "grad_norm": 0.473066508769989, "learning_rate": 1.987878787878788e-06, "loss": -0.0, "num_tokens": 5993422.0, "reward": 0.9981163740158081, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981163740158081, "reward_meter_std": 1.9677801901707426e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.96640412468696e-05, "reward_total_composite_mean": 0.9981163740158081, "reward_total_composite_std": 1.9677801901707426e-05, "reward_total_mean": 0.9981163740158081, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981163740158081, "rewards/meter/std": 1.9677801901707426e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981163740158081, "rewards/total_composite/std": 1.9677801901707426e-05, "sampling/importance_sampling_ratio/max": 1.1160743236541748, "sampling/importance_sampling_ratio/mean": 1.000008225440979, "sampling/importance_sampling_ratio/min": 0.6404457092285156, "sampling/sampling_logp_difference/max": 0.4455909729003906, "sampling/sampling_logp_difference/mean": 0.004463972058147192, "step": 2645 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.0018622480711201206, "epoch": 0.10627786480298831, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.984848484848485e-06, "loss": 0.0, "num_tokens": 5994830.0, "reward": 0.9943599104881287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9943599104881287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.003865361213684, "sampling/importance_sampling_ratio/mean": 1.0002386569976807, "sampling/importance_sampling_ratio/min": 0.999919056892395, "sampling/sampling_logp_difference/max": 0.0038578114472329617, "sampling/sampling_logp_difference/mean": 0.0002391562593402341, "step": 2646 }, { "clip_ratio/high_max": 0.023607419105246663, "clip_ratio/high_mean": 0.023607419105246663, "clip_ratio/low_mean": 0.006954187294468284, "clip_ratio/low_min": 0.006954187294468284, "clip_ratio/region_mean": 0.030561606399714947, "completions/clipped_ratio": 0.0, "completions/max_length": 199.0, "completions/max_terminated_length": 199.0, "completions/mean_length": 192.375, "completions/mean_terminated_length": 192.375, "completions/min_length": 185.0, "completions/min_terminated_length": 185.0, "entropy": 0.3521938771009445, "epoch": 0.10631803028477327, "frac_reward_zero_std": 0.0, "grad_norm": 2.459423303604126, "learning_rate": 1.981818181818182e-06, "loss": 0.0201, "num_tokens": 5997841.0, "reward": 0.9708031415939331, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985500574111938, "reward_meter_std": 0.0007222077692858875, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05117259919643402, "reward_total_composite_mean": 0.9708031415939331, "reward_total_composite_std": 0.05117257311940193, "reward_total_mean": 0.9708031415939331, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985500574111938, "rewards/meter/std": 0.0007222077692858875, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9708031415939331, "rewards/total_composite/std": 0.05117257311940193, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0101455450057983, "sampling/importance_sampling_ratio/min": 0.30152130126953125, "sampling/sampling_logp_difference/max": 1.1989145278930664, "sampling/sampling_logp_difference/mean": 0.04225311800837517, "step": 2647 }, { "clip_ratio/high_max": 0.01428450463572517, "clip_ratio/high_mean": 0.01428450463572517, "clip_ratio/low_mean": 0.000961538462433964, "clip_ratio/low_min": 0.000961538462433964, "clip_ratio/region_mean": 0.015246043098159134, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 131.25, "completions/mean_terminated_length": 131.25, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.08133039530366659, "epoch": 0.10635819576655822, "frac_reward_zero_std": 0.0, "grad_norm": 2.53367280960083, "learning_rate": 1.978787878787879e-06, "loss": -0.0035, "num_tokens": 6000235.0, "reward": 0.9814149141311646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992580413818359, "reward_meter_std": 5.2671864978037775e-05, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.050484977662563324, "reward_total_composite_mean": 0.9814149141311646, "reward_total_composite_std": 0.050484977662563324, "reward_total_mean": 0.9814149141311646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992580413818359, "rewards/meter/std": 5.2671864978037775e-05, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9814149141311646, "rewards/total_composite/std": 0.050484977662563324, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9994350075721741, "sampling/importance_sampling_ratio/min": 0.1773962378501892, "sampling/sampling_logp_difference/max": 1.7293694019317627, "sampling/sampling_logp_difference/mean": 0.014524316415190697, "step": 2648 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.004253393795806915, "epoch": 0.10639836124834318, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.975757575757576e-06, "loss": 0.0, "num_tokens": 6002123.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0139360427856445, "sampling/importance_sampling_ratio/mean": 1.0002467632293701, "sampling/importance_sampling_ratio/min": 0.9833143353462219, "sampling/sampling_logp_difference/max": 0.016826428472995758, "sampling/sampling_logp_difference/mean": 0.00039579044096171856, "step": 2649 }, { "clip_ratio/high_max": 0.007519222097471356, "clip_ratio/high_mean": 0.007519222097471356, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.007519222097471356, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.036395941860973835, "epoch": 0.10643852673012813, "frac_reward_zero_std": 0.0, "grad_norm": 0.25559931993484497, "learning_rate": 1.9727272727272727e-06, "loss": 0.0008, "num_tokens": 6003906.0, "reward": 0.9981404542922974, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981404542922974, "reward_meter_std": 1.4544022633344866e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.4552377251675352e-05, "reward_total_composite_mean": 0.9981404542922974, "reward_total_composite_std": 1.4544022633344866e-05, "reward_total_mean": 0.9981404542922974, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981404542922974, "rewards/meter/std": 1.4544022633344866e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981404542922974, "rewards/total_composite/std": 1.4544022633344866e-05, "sampling/importance_sampling_ratio/max": 1.3914573192596436, "sampling/importance_sampling_ratio/mean": 0.9998576641082764, "sampling/importance_sampling_ratio/min": 0.5241148471832275, "sampling/sampling_logp_difference/max": 0.6460444927215576, "sampling/sampling_logp_difference/mean": 0.005564166698604822, "step": 2650 }, { "epoch": 0.10643852673012813, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/max_length": 391.2307692307692, "eval_completions/max_terminated_length": 385.53846153846155, "eval_completions/mean_length": 202.45192307692307, "eval_completions/mean_terminated_length": 199.20054978590744, "eval_completions/min_length": 61.0, "eval_completions/min_terminated_length": 61.0, "eval_entropy": 0.33797027514531064, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 6003906.0, "eval_reward": 0.7124902972808251, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.9371546369332534, "eval_reward_count_adherence_std": 0.08564070440255679, "eval_reward_meter_mean": 0.8157925880872287, "eval_reward_meter_std": 0.30426162918313193, "eval_reward_repeat_penalty_mean": 0.9287464526983408, "eval_reward_repeat_penalty_std": 0.09337464204201332, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7124902972808251, "eval_reward_total_composite_std": 0.31389044053279436, "eval_reward_total_mean": 0.7124902972808251, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.9371546369332534, "eval_rewards/count_adherence/std": 0.08564070440255679, "eval_rewards/meter/mean": 0.8157925880872287, "eval_rewards/meter/std": 0.30426162918313193, "eval_rewards/repeat_penalty/mean": 0.9287464526983408, "eval_rewards/repeat_penalty/std": 0.09337464204201332, "eval_rewards/total_composite/mean": 0.7124902972808251, "eval_rewards/total_composite/std": 0.31389044053279436, "eval_runtime": 73.4017, "eval_samples_per_second": 1.417, "eval_sampling/importance_sampling_ratio/max": 1.4974250793457031, "eval_sampling/importance_sampling_ratio/mean": 1.0089522416775043, "eval_sampling/importance_sampling_ratio/min": 0.3371729736144726, "eval_sampling/sampling_logp_difference/max": 1.0985317963820238, "eval_sampling/sampling_logp_difference/mean": 0.03128652188640375, "eval_steps_per_second": 0.177, "step": 2650 }, { "clip_ratio/high_max": 0.005865102633833885, "clip_ratio/high_mean": 0.005865102633833885, "clip_ratio/low_mean": 0.024150803219527006, "clip_ratio/low_min": 0.024150803219527006, "clip_ratio/region_mean": 0.03001590585336089, "completions/clipped_ratio": 0.0, "completions/max_length": 341.0, "completions/max_terminated_length": 341.0, "completions/mean_length": 316.625, "completions/mean_terminated_length": 316.625, "completions/min_length": 300.0, "completions/min_terminated_length": 300.0, "entropy": 0.3517002984881401, "epoch": 0.10647869221191308, "frac_reward_zero_std": 0.0, "grad_norm": 2.0326192378997803, "learning_rate": 1.96969696969697e-06, "loss": -0.0237, "num_tokens": 6008199.0, "reward": 0.892192542552948, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.930555522441864, "reward_count_adherence_std": 0.05750546231865883, "reward_meter_mean": 0.9982768297195435, "reward_meter_std": 0.0005552266375161707, "reward_repeat_penalty_mean": 0.9622548818588257, "reward_repeat_penalty_std": 0.05441712588071823, "reward_std": 0.047268956899642944, "reward_total_composite_mean": 0.892192542552948, "reward_total_composite_std": 0.047268956899642944, "reward_total_mean": 0.892192542552948, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.930555522441864, "rewards/count_adherence/std": 0.05750546231865883, "rewards/meter/mean": 0.9982768297195435, "rewards/meter/std": 0.0005552266375161707, "rewards/repeat_penalty/mean": 0.9622548818588257, "rewards/repeat_penalty/std": 0.05441712588071823, "rewards/total_composite/mean": 0.892192542552948, "rewards/total_composite/std": 0.047268956899642944, "sampling/importance_sampling_ratio/max": 1.9213883876800537, "sampling/importance_sampling_ratio/mean": 1.0096842050552368, "sampling/importance_sampling_ratio/min": 0.11912889778614044, "sampling/sampling_logp_difference/max": 2.127549171447754, "sampling/sampling_logp_difference/mean": 0.04812028631567955, "step": 2651 }, { "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/low_mean": 0.004389182257000357, "clip_ratio/low_min": 0.004389182257000357, "clip_ratio/region_mean": 0.006125293381046504, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 142.125, "completions/mean_terminated_length": 142.125, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.09845188539475203, "epoch": 0.10651885769369804, "frac_reward_zero_std": 0.0, "grad_norm": 2.323451519012451, "learning_rate": 1.9666666666666668e-06, "loss": -0.004, "num_tokens": 6010904.0, "reward": 0.874354362487793, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992616176605225, "reward_meter_std": 9.245655382983387e-05, "reward_repeat_penalty_mean": 0.875, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.050480592995882034, "reward_total_composite_mean": 0.874354362487793, "reward_total_composite_std": 0.050480589270591736, "reward_total_mean": 0.874354362487793, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992616176605225, "rewards/meter/std": 9.245655382983387e-05, "rewards/repeat_penalty/mean": 0.875, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.874354362487793, "rewards/total_composite/std": 0.050480589270591736, "sampling/importance_sampling_ratio/max": 1.3792026042938232, "sampling/importance_sampling_ratio/mean": 1.0047122240066528, "sampling/importance_sampling_ratio/min": 0.3548658490180969, "sampling/sampling_logp_difference/max": 1.036015510559082, "sampling/sampling_logp_difference/mean": 0.01351867988705635, "step": 2652 }, { "clip_ratio/high_max": 0.02485937112942338, "clip_ratio/high_mean": 0.02485937112942338, "clip_ratio/low_mean": 0.0025445292703807354, "clip_ratio/low_min": 0.0025445292703807354, "clip_ratio/region_mean": 0.027403900399804115, "completions/clipped_ratio": 0.0, "completions/max_length": 425.0, "completions/max_terminated_length": 425.0, "completions/mean_length": 414.75, "completions/mean_terminated_length": 414.75, "completions/min_length": 393.0, "completions/min_terminated_length": 393.0, "entropy": 0.3583802431821823, "epoch": 0.10655902317548299, "frac_reward_zero_std": 0.0, "grad_norm": 1.9425232410430908, "learning_rate": 1.9636363636363636e-06, "loss": -0.0161, "num_tokens": 6015870.0, "reward": 0.7432065010070801, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7767857313156128, "reward_count_adherence_std": 0.02525380253791809, "reward_meter_mean": 0.9989011883735657, "reward_meter_std": 0.0002041146653937176, "reward_repeat_penalty_mean": 0.9577068090438843, "reward_repeat_penalty_std": 0.017178824171423912, "reward_std": 0.03013017773628235, "reward_total_composite_mean": 0.7432065010070801, "reward_total_composite_std": 0.030130181461572647, "reward_total_mean": 0.7432065010070801, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7767857313156128, "rewards/count_adherence/std": 0.02525380253791809, "rewards/meter/mean": 0.9989011883735657, "rewards/meter/std": 0.0002041146653937176, "rewards/repeat_penalty/mean": 0.9577068090438843, "rewards/repeat_penalty/std": 0.017178824171423912, "rewards/total_composite/mean": 0.7432065010070801, "rewards/total_composite/std": 0.030130181461572647, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0107157230377197, "sampling/importance_sampling_ratio/min": 0.21359705924987793, "sampling/sampling_logp_difference/max": 1.5436639785766602, "sampling/sampling_logp_difference/mean": 0.04433398321270943, "step": 2653 }, { "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/region_mean": 0.0035218254197388887, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 72.625, "completions/mean_terminated_length": 72.625, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.06974478112533689, "epoch": 0.10659918865726795, "frac_reward_zero_std": 0.0, "grad_norm": 1.9117748737335205, "learning_rate": 1.960606060606061e-06, "loss": -0.0037, "num_tokens": 6017843.0, "reward": 0.9993941783905029, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993941783905029, "reward_meter_std": 0.00015994130808394402, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001599277020432055, "reward_total_composite_mean": 0.9993941783905029, "reward_total_composite_std": 0.00015994130808394402, "reward_total_mean": 0.9993941783905029, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993941783905029, "rewards/meter/std": 0.00015994130808394402, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993941783905029, "rewards/total_composite/std": 0.00015994130808394402, "sampling/importance_sampling_ratio/max": 1.5019214153289795, "sampling/importance_sampling_ratio/mean": 1.0008054971694946, "sampling/importance_sampling_ratio/min": 0.4956419765949249, "sampling/sampling_logp_difference/max": 0.7019014358520508, "sampling/sampling_logp_difference/mean": 0.01229790784418583, "step": 2654 }, { "clip_ratio/high_max": 0.017125858925282955, "clip_ratio/high_mean": 0.017125858925282955, "clip_ratio/low_mean": 0.024847375229001045, "clip_ratio/low_min": 0.024847375229001045, "clip_ratio/region_mean": 0.041973234154284, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.25, "completions/mean_terminated_length": 71.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.4243501238524914, "epoch": 0.1066393541390529, "frac_reward_zero_std": 0.0, "grad_norm": 9.298754692077637, "learning_rate": 1.9575757575757577e-06, "loss": -0.014, "num_tokens": 6019661.0, "reward": 0.9782952070236206, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9782952070236206, "reward_meter_std": 0.013127398677170277, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01312740333378315, "reward_total_composite_mean": 0.9782952070236206, "reward_total_composite_std": 0.013127398677170277, "reward_total_mean": 0.9782952070236206, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9782952070236206, "rewards/meter/std": 0.013127398677170277, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9782952070236206, "rewards/total_composite/std": 0.013127398677170277, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0034326314926147, "sampling/importance_sampling_ratio/min": 0.08677612990140915, "sampling/sampling_logp_difference/max": 2.4444236755371094, "sampling/sampling_logp_difference/mean": 0.06088637933135033, "step": 2655 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.0018939394503831863, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.029891937039792538, "epoch": 0.10667951962083785, "frac_reward_zero_std": 0.0, "grad_norm": 0.48045873641967773, "learning_rate": 1.954545454545455e-06, "loss": 0.0, "num_tokens": 6021436.0, "reward": 0.9981427788734436, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981427788734436, "reward_meter_std": 1.307969432673417e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.3079693417239469e-05, "reward_total_composite_mean": 0.9981427788734436, "reward_total_composite_std": 1.307969432673417e-05, "reward_total_mean": 0.9981427788734436, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981427788734436, "rewards/meter/std": 1.307969432673417e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981427788734436, "rewards/total_composite/std": 1.307969432673417e-05, "sampling/importance_sampling_ratio/max": 1.6235767602920532, "sampling/importance_sampling_ratio/mean": 1.0033403635025024, "sampling/importance_sampling_ratio/min": 0.5827525854110718, "sampling/sampling_logp_difference/max": 0.5399925708770752, "sampling/sampling_logp_difference/mean": 0.004913026466965675, "step": 2656 }, { "clip_ratio/high_max": 0.031580700539052486, "clip_ratio/high_mean": 0.031580700539052486, "clip_ratio/low_mean": 0.0074352502124384046, "clip_ratio/low_min": 0.0074352502124384046, "clip_ratio/region_mean": 0.03901595075149089, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.125, "completions/mean_terminated_length": 70.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.44630854949355125, "epoch": 0.10671968510262281, "frac_reward_zero_std": 0.0, "grad_norm": 6.030494213104248, "learning_rate": 1.9515151515151518e-06, "loss": 0.0029, "num_tokens": 6023229.0, "reward": 0.9570232629776001, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9570232629776001, "reward_meter_std": 0.08651266247034073, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.08651266247034073, "reward_total_composite_mean": 0.9570232629776001, "reward_total_composite_std": 0.08651266247034073, "reward_total_mean": 0.9570232629776001, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9570232629776001, "rewards/meter/std": 0.08651266247034073, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9570232629776001, "rewards/total_composite/std": 0.08651266247034073, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0104464292526245, "sampling/importance_sampling_ratio/min": 0.3117186725139618, "sampling/sampling_logp_difference/max": 1.165654182434082, "sampling/sampling_logp_difference/mean": 0.05514438450336456, "step": 2657 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0046398533741012216, "epoch": 0.10675985058440776, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.9484848484848486e-06, "loss": 0.0, "num_tokens": 6025013.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0096546411514282, "sampling/importance_sampling_ratio/mean": 1.0001012086868286, "sampling/importance_sampling_ratio/min": 0.9832371473312378, "sampling/sampling_logp_difference/max": 0.01690489798784256, "sampling/sampling_logp_difference/mean": 0.00031924626091495156, "step": 2658 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.002239607958472334, "epoch": 0.10680001606619272, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.945454545454546e-06, "loss": 0.0, "num_tokens": 6026461.0, "reward": 0.9943599104881287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9943599104881287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0075657367706299, "sampling/importance_sampling_ratio/mean": 1.0003173351287842, "sampling/importance_sampling_ratio/min": 0.9997876882553101, "sampling/sampling_logp_difference/max": 0.00753726065158844, "sampling/sampling_logp_difference/mean": 0.00031867477810010314, "step": 2659 }, { "clip_ratio/high_max": 0.03413416654802859, "clip_ratio/high_mean": 0.03413416654802859, "clip_ratio/low_mean": 0.0330266528762877, "clip_ratio/low_min": 0.0330266528762877, "clip_ratio/region_mean": 0.06716081942431629, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.0, "completions/mean_terminated_length": 65.0, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.42190510779619217, "epoch": 0.10684018154797767, "frac_reward_zero_std": 0.0, "grad_norm": 11.615287780761719, "learning_rate": 1.9424242424242427e-06, "loss": 0.0104, "num_tokens": 6028445.0, "reward": 0.9398962259292603, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9398962259292603, "reward_meter_std": 0.01433228887617588, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014332284219563007, "reward_total_composite_mean": 0.9398962259292603, "reward_total_composite_std": 0.01433228887617588, "reward_total_mean": 0.9398962259292603, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9398962259292603, "rewards/meter/std": 0.01433228887617588, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9398962259292603, "rewards/total_composite/std": 0.01433228887617588, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0091454982757568, "sampling/importance_sampling_ratio/min": 0.22292140126228333, "sampling/sampling_logp_difference/max": 2.166830062866211, "sampling/sampling_logp_difference/mean": 0.0898202583193779, "step": 2660 }, { "clip_ratio/high_max": 0.008196720853447914, "clip_ratio/high_mean": 0.008196720853447914, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/region_mean": 0.012582685798406601, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 60.5, "completions/mean_terminated_length": 60.5, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.053432762157171965, "epoch": 0.10688034702976262, "frac_reward_zero_std": 0.0, "grad_norm": 3.7002193927764893, "learning_rate": 1.9393939393939395e-06, "loss": -0.0063, "num_tokens": 6030265.0, "reward": 0.9970978498458862, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970978498458862, "reward_meter_std": 0.00044736076961271465, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004473589942790568, "reward_total_composite_mean": 0.9970978498458862, "reward_total_composite_std": 0.00044736076961271465, "reward_total_mean": 0.9970978498458862, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970978498458862, "rewards/meter/std": 0.00044736076961271465, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970978498458862, "rewards/total_composite/std": 0.00044736076961271465, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0017904043197632, "sampling/importance_sampling_ratio/min": 0.642192542552948, "sampling/sampling_logp_difference/max": 0.8053483963012695, "sampling/sampling_logp_difference/mean": 0.011318504810333252, "step": 2661 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0019165669655194506, "epoch": 0.10692051251154758, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.9363636363636363e-06, "loss": 0.0, "num_tokens": 6031961.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0020672082901, "sampling/importance_sampling_ratio/mean": 1.0002366304397583, "sampling/importance_sampling_ratio/min": 0.9995288252830505, "sampling/sampling_logp_difference/max": 0.002065029926598072, "sampling/sampling_logp_difference/mean": 0.0002394376788288355, "step": 2662 }, { "clip_ratio/high_max": 0.008197940303944051, "clip_ratio/high_mean": 0.008197940303944051, "clip_ratio/low_mean": 0.007012551999650896, "clip_ratio/low_min": 0.007012551999650896, "clip_ratio/region_mean": 0.015210492303594947, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 90.625, "completions/mean_terminated_length": 90.625, "completions/min_length": 85.0, "completions/min_terminated_length": 85.0, "entropy": 0.10070738103240728, "epoch": 0.10696067799333253, "frac_reward_zero_std": 0.0, "grad_norm": 3.3872175216674805, "learning_rate": 1.9333333333333336e-06, "loss": -0.0089, "num_tokens": 6034070.0, "reward": 0.9972575306892395, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972575306892395, "reward_meter_std": 0.0004964250256307423, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004964345716871321, "reward_total_composite_mean": 0.9972575306892395, "reward_total_composite_std": 0.0004964250256307423, "reward_total_mean": 0.9972575306892395, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972575306892395, "rewards/meter/std": 0.0004964250256307423, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972575306892395, "rewards/total_composite/std": 0.0004964250256307423, "sampling/importance_sampling_ratio/max": 1.9014232158660889, "sampling/importance_sampling_ratio/mean": 0.9998160600662231, "sampling/importance_sampling_ratio/min": 0.31687599420547485, "sampling/sampling_logp_difference/max": 1.149244785308838, "sampling/sampling_logp_difference/mean": 0.02105124667286873, "step": 2663 }, { "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/low_mean": 0.014288630336523056, "clip_ratio/low_min": 0.014288630336523056, "clip_ratio/region_mean": 0.016024741460569203, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 70.5, "completions/mean_terminated_length": 70.5, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.10792513377964497, "epoch": 0.10700084347511749, "frac_reward_zero_std": 0.0, "grad_norm": 1.5907872915267944, "learning_rate": 1.9303030303030304e-06, "loss": -0.0067, "num_tokens": 6035802.0, "reward": 0.9993501305580139, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993501305580139, "reward_meter_std": 0.0001369249657727778, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001369154779240489, "reward_total_composite_mean": 0.9993501305580139, "reward_total_composite_std": 0.0001369249657727778, "reward_total_mean": 0.9993501305580139, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993501305580139, "rewards/meter/std": 0.0001369249657727778, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993501305580139, "rewards/total_composite/std": 0.0001369249657727778, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007887363433838, "sampling/importance_sampling_ratio/min": 0.4110756516456604, "sampling/sampling_logp_difference/max": 0.8889780044555664, "sampling/sampling_logp_difference/mean": 0.019528387114405632, "step": 2664 }, { "clip_ratio/high_max": 0.044199546333402395, "clip_ratio/high_mean": 0.044199546333402395, "clip_ratio/low_mean": 0.012102970853447914, "clip_ratio/low_min": 0.012102970853447914, "clip_ratio/region_mean": 0.05630251718685031, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 64.5, "completions/mean_terminated_length": 64.5, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.25452338345348835, "epoch": 0.10704100895690244, "frac_reward_zero_std": 0.0, "grad_norm": 12.907752990722656, "learning_rate": 1.9272727272727273e-06, "loss": -0.0018, "num_tokens": 6037606.0, "reward": 0.8035573959350586, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8035573959350586, "reward_meter_std": 0.2991119921207428, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2991119921207428, "reward_total_composite_mean": 0.8035573959350586, "reward_total_composite_std": 0.2991119921207428, "reward_total_mean": 0.8035573959350586, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8035573959350586, "rewards/meter/std": 0.2991119921207428, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8035573959350586, "rewards/total_composite/std": 0.2991119921207428, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.003564715385437, "sampling/importance_sampling_ratio/min": 0.14605195820331573, "sampling/sampling_logp_difference/max": 1.923792839050293, "sampling/sampling_logp_difference/mean": 0.06501687318086624, "step": 2665 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.002076218879665248, "epoch": 0.1070811744386874, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.924242424242424e-06, "loss": 0.0, "num_tokens": 6039238.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0026098489761353, "sampling/importance_sampling_ratio/mean": 1.000278115272522, "sampling/importance_sampling_ratio/min": 0.9999665021896362, "sampling/sampling_logp_difference/max": 0.00260642496868968, "sampling/sampling_logp_difference/mean": 0.00027814743225462735, "step": 2666 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0038265305338427424, "clip_ratio/low_min": 0.0038265305338427424, "clip_ratio/region_mean": 0.0038265305338427424, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.021847927011549473, "epoch": 0.10712133992047235, "frac_reward_zero_std": 0.0, "grad_norm": 0.08655895292758942, "learning_rate": 1.9212121212121213e-06, "loss": 0.0, "num_tokens": 6041406.0, "reward": 0.9993359446525574, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993359446525574, "reward_meter_std": 2.59008470493427e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.6060056370624807e-06, "reward_total_composite_mean": 0.9993359446525574, "reward_total_composite_std": 2.59008470493427e-06, "reward_total_mean": 0.9993359446525574, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993359446525574, "rewards/meter/std": 2.59008470493427e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993359446525574, "rewards/total_composite/std": 2.59008470493427e-06, "sampling/importance_sampling_ratio/max": 1.1448330879211426, "sampling/importance_sampling_ratio/mean": 1.0002384185791016, "sampling/importance_sampling_ratio/min": 0.74644535779953, "sampling/sampling_logp_difference/max": 0.29243287444114685, "sampling/sampling_logp_difference/mean": 0.0031785282772034407, "step": 2667 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 36.5, "completions/mean_terminated_length": 36.5, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.03415517252869904, "epoch": 0.1071615054022573, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.918181818181818e-06, "loss": 0.0, "num_tokens": 6043074.0, "reward": 0.9996045231819153, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9996045231819153, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.073022723197937, "sampling/importance_sampling_ratio/mean": 0.9983490109443665, "sampling/importance_sampling_ratio/min": 0.44583994150161743, "sampling/sampling_logp_difference/max": 0.8077952861785889, "sampling/sampling_logp_difference/mean": 0.005902368109673262, "step": 2668 }, { "clip_ratio/high_max": 0.041350313229486346, "clip_ratio/high_mean": 0.041350313229486346, "clip_ratio/low_mean": 0.006191797088831663, "clip_ratio/low_min": 0.006191797088831663, "clip_ratio/region_mean": 0.04754211031831801, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 102.625, "completions/mean_terminated_length": 102.625, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.39514047652482986, "epoch": 0.10720167088404225, "frac_reward_zero_std": 0.0, "grad_norm": 5.236100673675537, "learning_rate": 1.9151515151515154e-06, "loss": -0.0093, "num_tokens": 6045151.0, "reward": 0.9903435707092285, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9903435707092285, "reward_meter_std": 0.007074963301420212, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00707497913390398, "reward_total_composite_mean": 0.9903435707092285, "reward_total_composite_std": 0.007074963301420212, "reward_total_mean": 0.9903435707092285, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9903435707092285, "rewards/meter/std": 0.007074963301420212, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9903435707092285, "rewards/total_composite/std": 0.007074963301420212, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001228928565979, "sampling/importance_sampling_ratio/min": 0.2917172312736511, "sampling/sampling_logp_difference/max": 1.2319703102111816, "sampling/sampling_logp_difference/mean": 0.04934058338403702, "step": 2669 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018656715983524919, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.016892601503059268, "epoch": 0.10724183636582721, "frac_reward_zero_std": 0.0, "grad_norm": 0.37225642800331116, "learning_rate": 1.9121212121212123e-06, "loss": -0.0, "num_tokens": 6046947.0, "reward": 0.9981483817100525, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981483817100525, "reward_meter_std": 1.0958180610032286e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.0958180610032286e-05, "reward_total_composite_mean": 0.9981483817100525, "reward_total_composite_std": 1.0958180610032286e-05, "reward_total_mean": 0.9981483817100525, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981483817100525, "rewards/meter/std": 1.0958180610032286e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981483817100525, "rewards/total_composite/std": 1.0958180610032286e-05, "sampling/importance_sampling_ratio/max": 1.4169889688491821, "sampling/importance_sampling_ratio/mean": 1.0005658864974976, "sampling/importance_sampling_ratio/min": 0.1989072561264038, "sampling/sampling_logp_difference/max": 1.6149165630340576, "sampling/sampling_logp_difference/mean": 0.006186656653881073, "step": 2670 }, { "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/low_mean": 0.014549180399626493, "clip_ratio/low_min": 0.014549180399626493, "clip_ratio/region_mean": 0.01864754082635045, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 60.75, "completions/mean_terminated_length": 60.75, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.07036425499245524, "epoch": 0.10728200184761216, "frac_reward_zero_std": 0.0, "grad_norm": 7.428125381469727, "learning_rate": 1.9090909090909095e-06, "loss": 0.0007, "num_tokens": 6048617.0, "reward": 0.9970404505729675, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970404505729675, "reward_meter_std": 0.00046239994117058814, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004624156281352043, "reward_total_composite_mean": 0.9970404505729675, "reward_total_composite_std": 0.00046239994117058814, "reward_total_mean": 0.9970404505729675, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970404505729675, "rewards/meter/std": 0.00046239994117058814, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970404505729675, "rewards/total_composite/std": 0.00046239994117058814, "sampling/importance_sampling_ratio/max": 1.6209132671356201, "sampling/importance_sampling_ratio/mean": 1.0006794929504395, "sampling/importance_sampling_ratio/min": 0.4077909588813782, "sampling/sampling_logp_difference/max": 0.8970005512237549, "sampling/sampling_logp_difference/mean": 0.014924735762178898, "step": 2671 }, { "clip_ratio/high_max": 0.022553268587216735, "clip_ratio/high_mean": 0.022553268587216735, "clip_ratio/low_mean": 0.0011718750465661287, "clip_ratio/low_min": 0.0011718750465661287, "clip_ratio/region_mean": 0.023725143633782864, "completions/clipped_ratio": 0.0, "completions/max_length": 328.0, "completions/max_terminated_length": 328.0, "completions/mean_length": 316.625, "completions/mean_terminated_length": 316.625, "completions/min_length": 309.0, "completions/min_terminated_length": 309.0, "entropy": 0.46788543090224266, "epoch": 0.10732216732939712, "frac_reward_zero_std": 0.0, "grad_norm": 6.031638145446777, "learning_rate": 1.9060606060606064e-06, "loss": 0.0098, "num_tokens": 6052766.0, "reward": 0.6005241870880127, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.8409091234207153, "reward_count_adherence_std": 0.04208274558186531, "reward_meter_mean": 0.9499752521514893, "reward_meter_std": 0.06226501986384392, "reward_repeat_penalty_mean": 0.8695175647735596, "reward_repeat_penalty_std": 0.058811455965042114, "reward_std": 0.24652279913425446, "reward_total_composite_mean": 0.6005241870880127, "reward_total_composite_std": 0.24652278423309326, "reward_total_mean": 0.6005241870880127, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.8409091234207153, "rewards/count_adherence/std": 0.04208274558186531, "rewards/meter/mean": 0.9499752521514893, "rewards/meter/std": 0.06226501986384392, "rewards/repeat_penalty/mean": 0.8695175647735596, "rewards/repeat_penalty/std": 0.058811455965042114, "rewards/total_composite/mean": 0.6005241870880127, "rewards/total_composite/std": 0.24652278423309326, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0106778144836426, "sampling/importance_sampling_ratio/min": 0.19964751601219177, "sampling/sampling_logp_difference/max": 1.6112018823623657, "sampling/sampling_logp_difference/mean": 0.04524172842502594, "step": 2672 }, { "clip_ratio/high_max": 0.006329114083200693, "clip_ratio/high_mean": 0.006329114083200693, "clip_ratio/low_mean": 0.00795196380931884, "clip_ratio/low_min": 0.00795196380931884, "clip_ratio/region_mean": 0.014281077892519534, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 79.125, "completions/mean_terminated_length": 79.125, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.12998810317367315, "epoch": 0.10736233281118207, "frac_reward_zero_std": 0.0, "grad_norm": 1.5127347707748413, "learning_rate": 1.9030303030303032e-06, "loss": -0.0033, "num_tokens": 6054615.0, "reward": 0.99886155128479, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99886155128479, "reward_meter_std": 0.00016637769294902682, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00016637658700346947, "reward_total_composite_mean": 0.99886155128479, "reward_total_composite_std": 0.00016637769294902682, "reward_total_mean": 0.99886155128479, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99886155128479, "rewards/meter/std": 0.00016637769294902682, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99886155128479, "rewards/total_composite/std": 0.00016637769294902682, "sampling/importance_sampling_ratio/max": 1.7465615272521973, "sampling/importance_sampling_ratio/mean": 1.0049008131027222, "sampling/importance_sampling_ratio/min": 0.43897879123687744, "sampling/sampling_logp_difference/max": 0.8233041763305664, "sampling/sampling_logp_difference/mean": 0.0236604493111372, "step": 2673 }, { "clip_ratio/high_max": 0.012176374206319451, "clip_ratio/high_mean": 0.012176374206319451, "clip_ratio/low_mean": 0.012097747880034149, "clip_ratio/low_min": 0.012097747880034149, "clip_ratio/region_mean": 0.0242741220863536, "completions/clipped_ratio": 0.0, "completions/max_length": 272.0, "completions/max_terminated_length": 272.0, "completions/mean_length": 268.375, "completions/mean_terminated_length": 268.375, "completions/min_length": 262.0, "completions/min_terminated_length": 262.0, "entropy": 0.33767288736999035, "epoch": 0.10740249829296702, "frac_reward_zero_std": 0.0, "grad_norm": 2.1606380939483643, "learning_rate": 1.9000000000000002e-06, "loss": 0.0073, "num_tokens": 6058386.0, "reward": 0.9412178993225098, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988313317298889, "reward_meter_std": 0.0005680599715560675, "reward_repeat_penalty_mean": 0.942307710647583, "reward_repeat_penalty_std": 0.054392825812101364, "reward_std": 0.0545286163687706, "reward_total_composite_mean": 0.9412178993225098, "reward_total_composite_std": 0.0545286163687706, "reward_total_mean": 0.9412178993225098, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988313317298889, "rewards/meter/std": 0.0005680599715560675, "rewards/repeat_penalty/mean": 0.942307710647583, "rewards/repeat_penalty/std": 0.054392825812101364, "rewards/total_composite/mean": 0.9412178993225098, "rewards/total_composite/std": 0.0545286163687706, "sampling/importance_sampling_ratio/max": 1.7729755640029907, "sampling/importance_sampling_ratio/mean": 1.0066032409667969, "sampling/importance_sampling_ratio/min": 0.23089425265789032, "sampling/sampling_logp_difference/max": 1.4657955169677734, "sampling/sampling_logp_difference/mean": 0.043261099606752396, "step": 2674 }, { "clip_ratio/high_max": 0.014709851937368512, "clip_ratio/high_mean": 0.014709851937368512, "clip_ratio/low_mean": 0.009543466847389936, "clip_ratio/low_min": 0.009543466847389936, "clip_ratio/region_mean": 0.02425331878475845, "completions/clipped_ratio": 0.0, "completions/max_length": 346.0, "completions/max_terminated_length": 346.0, "completions/mean_length": 340.75, "completions/mean_terminated_length": 340.75, "completions/min_length": 337.0, "completions/min_terminated_length": 337.0, "entropy": 0.3350350670516491, "epoch": 0.10744266377475198, "frac_reward_zero_std": 0.0, "grad_norm": 1.5513638257980347, "learning_rate": 1.896969696969697e-06, "loss": 0.0064, "num_tokens": 6062752.0, "reward": 0.8193838000297546, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985557794570923, "reward_meter_std": 0.0006976545555517077, "reward_repeat_penalty_mean": 0.9117647409439087, "reward_repeat_penalty_std": 0.05446000397205353, "reward_std": 0.04855942353606224, "reward_total_composite_mean": 0.8193838000297546, "reward_total_composite_std": 0.04855939745903015, "reward_total_mean": 0.8193838000297546, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985557794570923, "rewards/meter/std": 0.0006976545555517077, "rewards/repeat_penalty/mean": 0.9117647409439087, "rewards/repeat_penalty/std": 0.05446000397205353, "rewards/total_composite/mean": 0.8193838000297546, "rewards/total_composite/std": 0.04855939745903015, "sampling/importance_sampling_ratio/max": 1.8999286890029907, "sampling/importance_sampling_ratio/mean": 1.0090699195861816, "sampling/importance_sampling_ratio/min": 0.2906521260738373, "sampling/sampling_logp_difference/max": 1.2356281280517578, "sampling/sampling_logp_difference/mean": 0.040726643055677414, "step": 2675 }, { "clip_ratio/high_max": 0.0071428571827709675, "clip_ratio/high_mean": 0.0071428571827709675, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.010819327784702182, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 34.625, "completions/mean_terminated_length": 34.625, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.07462144270539284, "epoch": 0.10748282925653693, "frac_reward_zero_std": 0.0, "grad_norm": 3.5680060386657715, "learning_rate": 1.8939393939393941e-06, "loss": -0.0079, "num_tokens": 6064157.0, "reward": 0.9994817972183228, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994817972183228, "reward_meter_std": 0.0001635835214983672, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00016358528228010982, "reward_total_composite_mean": 0.9994817972183228, "reward_total_composite_std": 0.0001635835214983672, "reward_total_mean": 0.9994817972183228, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994817972183228, "rewards/meter/std": 0.0001635835214983672, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994817972183228, "rewards/total_composite/std": 0.0001635835214983672, "sampling/importance_sampling_ratio/max": 1.5120035409927368, "sampling/importance_sampling_ratio/mean": 1.0051720142364502, "sampling/importance_sampling_ratio/min": 0.49631646275520325, "sampling/sampling_logp_difference/max": 0.7005414962768555, "sampling/sampling_logp_difference/mean": 0.015038218349218369, "step": 2676 }, { "clip_ratio/high_max": 0.029550255741924047, "clip_ratio/high_mean": 0.029550255741924047, "clip_ratio/low_mean": 0.007593599148094654, "clip_ratio/low_min": 0.007593599148094654, "clip_ratio/region_mean": 0.0371438548900187, "completions/clipped_ratio": 0.0, "completions/max_length": 184.0, "completions/max_terminated_length": 184.0, "completions/mean_length": 175.5, "completions/mean_terminated_length": 175.5, "completions/min_length": 167.0, "completions/min_terminated_length": 167.0, "entropy": 0.19678651168942451, "epoch": 0.10752299473832189, "frac_reward_zero_std": 0.0, "grad_norm": 3.107592821121216, "learning_rate": 1.890909090909091e-06, "loss": 0.0237, "num_tokens": 6067073.0, "reward": 0.974378228187561, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970433712005615, "reward_meter_std": 0.0006079506129026413, "reward_repeat_penalty_mean": 0.9772727489471436, "reward_repeat_penalty_std": 0.04208271950483322, "reward_std": 0.041835810989141464, "reward_total_composite_mean": 0.974378228187561, "reward_total_composite_std": 0.041835807263851166, "reward_total_mean": 0.974378228187561, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970433712005615, "rewards/meter/std": 0.0006079506129026413, "rewards/repeat_penalty/mean": 0.9772727489471436, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.974378228187561, "rewards/total_composite/std": 0.041835807263851166, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0057967901229858, "sampling/importance_sampling_ratio/min": 0.135078564286232, "sampling/sampling_logp_difference/max": 2.001898765563965, "sampling/sampling_logp_difference/mean": 0.033983275294303894, "step": 2677 }, { "clip_ratio/high_max": 0.014815408736467361, "clip_ratio/high_mean": 0.014815408736467361, "clip_ratio/low_mean": 0.004108626628294587, "clip_ratio/low_min": 0.004108626628294587, "clip_ratio/region_mean": 0.01892403536476195, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 91.75, "completions/mean_terminated_length": 91.75, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.12407996132969856, "epoch": 0.10756316022010684, "frac_reward_zero_std": 0.0, "grad_norm": 2.5906176567077637, "learning_rate": 1.887878787878788e-06, "loss": -0.0038, "num_tokens": 6069055.0, "reward": 0.9974672198295593, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974672198295593, "reward_meter_std": 0.0002358894416829571, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00023587615578435361, "reward_total_composite_mean": 0.9974672198295593, "reward_total_composite_std": 0.0002358894416829571, "reward_total_mean": 0.9974672198295593, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974672198295593, "rewards/meter/std": 0.0002358894416829571, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974672198295593, "rewards/total_composite/std": 0.0002358894416829571, "sampling/importance_sampling_ratio/max": 1.6885485649108887, "sampling/importance_sampling_ratio/mean": 1.0014500617980957, "sampling/importance_sampling_ratio/min": 0.2142159342765808, "sampling/sampling_logp_difference/max": 1.5407707691192627, "sampling/sampling_logp_difference/mean": 0.025346064940094948, "step": 2678 }, { "clip_ratio/high_max": 0.021945541491732, "clip_ratio/high_mean": 0.021945541491732, "clip_ratio/low_mean": 0.007462686393409967, "clip_ratio/low_min": 0.007462686393409967, "clip_ratio/region_mean": 0.02940822788514197, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 67.625, "completions/mean_terminated_length": 67.625, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.2862533424049616, "epoch": 0.1076033257018918, "frac_reward_zero_std": 0.0, "grad_norm": 3.563011646270752, "learning_rate": 1.884848484848485e-06, "loss": 0.0065, "num_tokens": 6070908.0, "reward": 0.9295022487640381, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9295022487640381, "reward_meter_std": 0.1213918924331665, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.12139188498258591, "reward_total_composite_mean": 0.9295022487640381, "reward_total_composite_std": 0.1213918924331665, "reward_total_mean": 0.9295022487640381, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9295022487640381, "rewards/meter/std": 0.1213918924331665, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9295022487640381, "rewards/total_composite/std": 0.1213918924331665, "sampling/importance_sampling_ratio/max": 1.9054324626922607, "sampling/importance_sampling_ratio/mean": 1.0032422542572021, "sampling/importance_sampling_ratio/min": 0.31223487854003906, "sampling/sampling_logp_difference/max": 1.1639995574951172, "sampling/sampling_logp_difference/mean": 0.034729912877082825, "step": 2679 }, { "clip_ratio/high_max": 0.024966524448245764, "clip_ratio/high_mean": 0.024966524448245764, "clip_ratio/low_mean": 0.0015060240402817726, "clip_ratio/low_min": 0.0015060240402817726, "clip_ratio/region_mean": 0.026472548488527536, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 89.75, "completions/mean_terminated_length": 89.75, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.12778305634856224, "epoch": 0.10764349118367675, "frac_reward_zero_std": 0.0, "grad_norm": 8.263280868530273, "learning_rate": 1.8818181818181819e-06, "loss": -0.0195, "num_tokens": 6072994.0, "reward": 0.9924962520599365, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9924962520599365, "reward_meter_std": 0.013577880337834358, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01357787661254406, "reward_total_composite_mean": 0.9924962520599365, "reward_total_composite_std": 0.013577880337834358, "reward_total_mean": 0.9924962520599365, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9924962520599365, "rewards/meter/std": 0.013577880337834358, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924962520599365, "rewards/total_composite/std": 0.013577880337834358, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9978711605072021, "sampling/importance_sampling_ratio/min": 0.2512976825237274, "sampling/sampling_logp_difference/max": 1.3811171054840088, "sampling/sampling_logp_difference/mean": 0.030782943591475487, "step": 2680 }, { "clip_ratio/high_max": 0.004713488393463194, "clip_ratio/high_mean": 0.004713488393463194, "clip_ratio/low_mean": 0.006650691735558212, "clip_ratio/low_min": 0.006650691735558212, "clip_ratio/region_mean": 0.011364180129021406, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 132.125, "completions/mean_terminated_length": 132.125, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.058492994867265224, "epoch": 0.1076836566654617, "frac_reward_zero_std": 0.0, "grad_norm": 2.0896339416503906, "learning_rate": 1.878787878787879e-06, "loss": -0.0005, "num_tokens": 6075627.0, "reward": 0.9991657733917236, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991657733917236, "reward_meter_std": 0.0002873947087209672, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00028738551191054285, "reward_total_composite_mean": 0.9991657733917236, "reward_total_composite_std": 0.0002873947087209672, "reward_total_mean": 0.9991657733917236, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991657733917236, "rewards/meter/std": 0.0002873947087209672, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991657733917236, "rewards/total_composite/std": 0.0002873947087209672, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9993587136268616, "sampling/importance_sampling_ratio/min": 0.23843567073345184, "sampling/sampling_logp_difference/max": 1.4336557388305664, "sampling/sampling_logp_difference/mean": 0.015285594388842583, "step": 2681 }, { "clip_ratio/high_max": 0.010841309442184865, "clip_ratio/high_mean": 0.010841309442184865, "clip_ratio/low_mean": 0.021640333347022533, "clip_ratio/low_min": 0.021640333347022533, "clip_ratio/region_mean": 0.0324816427892074, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 92.625, "completions/mean_terminated_length": 92.625, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.1409629676491022, "epoch": 0.10772382214724666, "frac_reward_zero_std": 0.0, "grad_norm": 2.7295045852661133, "learning_rate": 1.8757575757575757e-06, "loss": 0.0047, "num_tokens": 6077792.0, "reward": 0.9971524477005005, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971524477005005, "reward_meter_std": 0.0005293982103466988, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005293957656249404, "reward_total_composite_mean": 0.9971524477005005, "reward_total_composite_std": 0.0005293982103466988, "reward_total_mean": 0.9971524477005005, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971524477005005, "rewards/meter/std": 0.0005293982103466988, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971524477005005, "rewards/total_composite/std": 0.0005293982103466988, "sampling/importance_sampling_ratio/max": 1.9131104946136475, "sampling/importance_sampling_ratio/mean": 1.0033948421478271, "sampling/importance_sampling_ratio/min": 0.24620838463306427, "sampling/sampling_logp_difference/max": 1.4015769958496094, "sampling/sampling_logp_difference/mean": 0.03304116055369377, "step": 2682 }, { "clip_ratio/high_max": 0.014336680993437767, "clip_ratio/high_mean": 0.014336680993437767, "clip_ratio/low_mean": 0.02360188332386315, "clip_ratio/low_min": 0.02360188332386315, "clip_ratio/region_mean": 0.037938564317300916, "completions/clipped_ratio": 0.0, "completions/max_length": 44.0, "completions/max_terminated_length": 44.0, "completions/mean_length": 42.75, "completions/mean_terminated_length": 42.75, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.28468130715191364, "epoch": 0.10776398762903161, "frac_reward_zero_std": 0.0, "grad_norm": 13.13068962097168, "learning_rate": 1.872727272727273e-06, "loss": -0.0069, "num_tokens": 6079318.0, "reward": 0.953346848487854, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.953346848487854, "reward_meter_std": 0.009581836871802807, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009581834077835083, "reward_total_composite_mean": 0.953346848487854, "reward_total_composite_std": 0.009581836871802807, "reward_total_mean": 0.953346848487854, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.953346848487854, "rewards/meter/std": 0.009581836871802807, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.953346848487854, "rewards/total_composite/std": 0.009581836871802807, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015714168548584, "sampling/importance_sampling_ratio/min": 0.3157352805137634, "sampling/sampling_logp_difference/max": 1.1528511047363281, "sampling/sampling_logp_difference/mean": 0.06227937340736389, "step": 2683 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 41.0, "completions/mean_terminated_length": 41.0, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.01764180068857968, "epoch": 0.10780415311081656, "frac_reward_zero_std": 0.0, "grad_norm": 0.11564414203166962, "learning_rate": 1.86969696969697e-06, "loss": -0.0002, "num_tokens": 6081094.0, "reward": 0.9985654354095459, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985654354095459, "reward_meter_std": 4.657397312257672e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.660256763600046e-06, "reward_total_composite_mean": 0.9985654354095459, "reward_total_composite_std": 4.657397312257672e-06, "reward_total_mean": 0.9985654354095459, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985654354095459, "rewards/meter/std": 4.657397312257672e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9985654354095459, "rewards/total_composite/std": 4.657397312257672e-06, "sampling/importance_sampling_ratio/max": 1.0496618747711182, "sampling/importance_sampling_ratio/mean": 1.0005881786346436, "sampling/importance_sampling_ratio/min": 0.6048609614372253, "sampling/sampling_logp_difference/max": 0.5027565956115723, "sampling/sampling_logp_difference/mean": 0.0033583450131118298, "step": 2684 }, { "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/low_mean": 0.004385964944958687, "clip_ratio/low_min": 0.004385964944958687, "clip_ratio/region_mean": 0.010533505585044622, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 60.5, "completions/mean_terminated_length": 60.5, "completions/min_length": 57.0, "completions/min_terminated_length": 57.0, "entropy": 0.06518607307225466, "epoch": 0.10784431859260152, "frac_reward_zero_std": 0.0, "grad_norm": 5.177779197692871, "learning_rate": 1.8666666666666669e-06, "loss": -0.0036, "num_tokens": 6082730.0, "reward": 0.9971067905426025, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971067905426025, "reward_meter_std": 0.0004163467965554446, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004163416160736233, "reward_total_composite_mean": 0.9971067905426025, "reward_total_composite_std": 0.0004163467965554446, "reward_total_mean": 0.9971067905426025, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971067905426025, "rewards/meter/std": 0.0004163467965554446, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971067905426025, "rewards/total_composite/std": 0.0004163467965554446, "sampling/importance_sampling_ratio/max": 1.7392834424972534, "sampling/importance_sampling_ratio/mean": 1.0009123086929321, "sampling/importance_sampling_ratio/min": 0.4041905701160431, "sampling/sampling_logp_difference/max": 0.9058688879013062, "sampling/sampling_logp_difference/mean": 0.014101680368185043, "step": 2685 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0043548993999138474, "epoch": 0.10788448407438647, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.863636363636364e-06, "loss": 0.0, "num_tokens": 6084138.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.010701298713684, "sampling/importance_sampling_ratio/mean": 0.999911904335022, "sampling/importance_sampling_ratio/min": 0.9723634123802185, "sampling/sampling_logp_difference/max": 0.028025653213262558, "sampling/sampling_logp_difference/mean": 0.0005363853415474296, "step": 2686 }, { "clip_ratio/high_max": 0.020963466726243496, "clip_ratio/high_mean": 0.020963466726243496, "clip_ratio/low_mean": 0.0056649468606337905, "clip_ratio/low_min": 0.0056649468606337905, "clip_ratio/region_mean": 0.026628413586877286, "completions/clipped_ratio": 0.0, "completions/max_length": 491.0, "completions/max_terminated_length": 491.0, "completions/mean_length": 472.75, "completions/mean_terminated_length": 472.75, "completions/min_length": 433.0, "completions/min_terminated_length": 433.0, "entropy": 0.35818639770150185, "epoch": 0.10792464955617143, "frac_reward_zero_std": 0.0, "grad_norm": 1.300607681274414, "learning_rate": 1.8606060606060607e-06, "loss": -0.0234, "num_tokens": 6090080.0, "reward": 0.7318886518478394, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7734375, "reward_count_adherence_std": 0.04650149121880531, "reward_meter_mean": 0.998869776725769, "reward_meter_std": 0.00026758189778774977, "reward_repeat_penalty_mean": 0.946873664855957, "reward_repeat_penalty_std": 0.06902104616165161, "reward_std": 0.07250804454088211, "reward_total_composite_mean": 0.7318886518478394, "reward_total_composite_std": 0.0725080594420433, "reward_total_mean": 0.7318886518478394, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7734375, "rewards/count_adherence/std": 0.04650149121880531, "rewards/meter/mean": 0.998869776725769, "rewards/meter/std": 0.00026758189778774977, "rewards/repeat_penalty/mean": 0.946873664855957, "rewards/repeat_penalty/std": 0.06902104616165161, "rewards/total_composite/mean": 0.7318886518478394, "rewards/total_composite/std": 0.0725080594420433, "sampling/importance_sampling_ratio/max": 1.9875471591949463, "sampling/importance_sampling_ratio/mean": 1.009418249130249, "sampling/importance_sampling_ratio/min": 0.1827991008758545, "sampling/sampling_logp_difference/max": 1.6993675231933594, "sampling/sampling_logp_difference/mean": 0.03919937461614609, "step": 2687 }, { "clip_ratio/high_max": 0.04208028828725219, "clip_ratio/high_mean": 0.04208028828725219, "clip_ratio/low_mean": 0.008036528481170535, "clip_ratio/low_min": 0.008036528481170535, "clip_ratio/region_mean": 0.05011681676842272, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 140.125, "completions/mean_terminated_length": 140.125, "completions/min_length": 137.0, "completions/min_terminated_length": 137.0, "entropy": 0.39379359781742096, "epoch": 0.10796481503795638, "frac_reward_zero_std": 0.0, "grad_norm": 3.9790732860565186, "learning_rate": 1.8575757575757578e-06, "loss": 0.0052, "num_tokens": 6092497.0, "reward": 0.9988832473754883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988832473754883, "reward_meter_std": 0.0008235845598392189, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008235886925831437, "reward_total_composite_mean": 0.9988832473754883, "reward_total_composite_std": 0.0008235845598392189, "reward_total_mean": 0.9988832473754883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988832473754883, "rewards/meter/std": 0.0008235845598392189, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988832473754883, "rewards/total_composite/std": 0.0008235845598392189, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.00178861618042, "sampling/importance_sampling_ratio/min": 0.09384826570749283, "sampling/sampling_logp_difference/max": 2.3660759925842285, "sampling/sampling_logp_difference/mean": 0.0495881512761116, "step": 2688 }, { "clip_ratio/high_max": 0.00855186500120908, "clip_ratio/high_mean": 0.00855186500120908, "clip_ratio/low_mean": 0.004699247889220715, "clip_ratio/low_min": 0.004699247889220715, "clip_ratio/region_mean": 0.013251112890429795, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 132.0, "completions/mean_terminated_length": 132.0, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.06141908373683691, "epoch": 0.10800498051974133, "frac_reward_zero_std": 0.0, "grad_norm": 3.656802177429199, "learning_rate": 1.8545454545454546e-06, "loss": 0.0078, "num_tokens": 6095097.0, "reward": 0.9636418223381042, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993307590484619, "reward_meter_std": 7.24185592844151e-05, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06610793620347977, "reward_total_composite_mean": 0.9636418223381042, "reward_total_composite_std": 0.06610792875289917, "reward_total_mean": 0.9636418223381042, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993307590484619, "rewards/meter/std": 7.24185592844151e-05, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9636418223381042, "rewards/total_composite/std": 0.06610792875289917, "sampling/importance_sampling_ratio/max": 1.8010199069976807, "sampling/importance_sampling_ratio/mean": 0.9992504715919495, "sampling/importance_sampling_ratio/min": 0.4471226632595062, "sampling/sampling_logp_difference/max": 0.804922342300415, "sampling/sampling_logp_difference/mean": 0.011180154979228973, "step": 2689 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.007268769608344883, "epoch": 0.10804514600152629, "frac_reward_zero_std": 0.0, "grad_norm": 0.08232353627681732, "learning_rate": 1.8515151515151517e-06, "loss": 0.0005, "num_tokens": 6097105.0, "reward": 0.9981546401977539, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981546401977539, "reward_meter_std": 6.638248123636004e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.6471684476709925e-06, "reward_total_composite_mean": 0.9981546401977539, "reward_total_composite_std": 6.638248123636004e-06, "reward_total_mean": 0.9981546401977539, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981546401977539, "rewards/meter/std": 6.638248123636004e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981546401977539, "rewards/total_composite/std": 6.638248123636004e-06, "sampling/importance_sampling_ratio/max": 1.0263193845748901, "sampling/importance_sampling_ratio/mean": 0.9995981454849243, "sampling/importance_sampling_ratio/min": 0.5742889642715454, "sampling/sampling_logp_difference/max": 0.5546226501464844, "sampling/sampling_logp_difference/mean": 0.0016760394209995866, "step": 2690 }, { "clip_ratio/high_max": 0.009009360568597913, "clip_ratio/high_mean": 0.009009360568597913, "clip_ratio/low_mean": 0.0015015150420367718, "clip_ratio/low_min": 0.0015015150420367718, "clip_ratio/region_mean": 0.010510875610634685, "completions/clipped_ratio": 0.0, "completions/max_length": 168.0, "completions/max_terminated_length": 168.0, "completions/mean_length": 166.375, "completions/mean_terminated_length": 166.375, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.07314449176192284, "epoch": 0.10808531148331124, "frac_reward_zero_std": 0.0, "grad_norm": 2.4885315895080566, "learning_rate": 1.8484848484848487e-06, "loss": 0.0031, "num_tokens": 6099876.0, "reward": 0.9715026617050171, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992619752883911, "reward_meter_std": 0.00024342280812561512, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05135667324066162, "reward_total_composite_mean": 0.9715026617050171, "reward_total_composite_std": 0.051356665790081024, "reward_total_mean": 0.9715026617050171, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992619752883911, "rewards/meter/std": 0.00024342280812561512, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9715026617050171, "rewards/total_composite/std": 0.051356665790081024, "sampling/importance_sampling_ratio/max": 1.893328070640564, "sampling/importance_sampling_ratio/mean": 1.0019655227661133, "sampling/importance_sampling_ratio/min": 0.24973411858081818, "sampling/sampling_logp_difference/max": 1.3873584270477295, "sampling/sampling_logp_difference/mean": 0.015284853056073189, "step": 2691 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0021011842327425256, "epoch": 0.1081254769650962, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.8454545454545455e-06, "loss": 0.0, "num_tokens": 6101556.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.004001498222351, "sampling/importance_sampling_ratio/mean": 1.0002787113189697, "sampling/importance_sampling_ratio/min": 1.0, "sampling/sampling_logp_difference/max": 0.003993465099483728, "sampling/sampling_logp_difference/mean": 0.0002783673407975584, "step": 2692 }, { "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/low_mean": 0.0038265305338427424, "clip_ratio/low_min": 0.0038265305338427424, "clip_ratio/region_mean": 0.006377550889737904, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.01480487291701138, "epoch": 0.10816564244688115, "frac_reward_zero_std": 0.0, "grad_norm": 1.4883558750152588, "learning_rate": 1.8424242424242426e-06, "loss": 0.0009, "num_tokens": 6103724.0, "reward": 0.9992461204528809, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992461204528809, "reward_meter_std": 0.00010268352343700826, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010269100312143564, "reward_total_composite_mean": 0.9992461204528809, "reward_total_composite_std": 0.00010268352343700826, "reward_total_mean": 0.9992461204528809, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992461204528809, "rewards/meter/std": 0.00010268352343700826, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992461204528809, "rewards/total_composite/std": 0.00010268352343700826, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0015344619750977, "sampling/importance_sampling_ratio/min": 0.6744450330734253, "sampling/sampling_logp_difference/max": 1.3028526306152344, "sampling/sampling_logp_difference/mean": 0.005682831630110741, "step": 2693 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0020191115036141127, "epoch": 0.1082058079286661, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.8393939393939394e-06, "loss": 0.0, "num_tokens": 6105620.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0060728788375854, "sampling/importance_sampling_ratio/mean": 1.000247597694397, "sampling/importance_sampling_ratio/min": 0.9999150037765503, "sampling/sampling_logp_difference/max": 0.006054490804672241, "sampling/sampling_logp_difference/mean": 0.0002477174566593021, "step": 2694 }, { "clip_ratio/high_max": 0.0434071971103549, "clip_ratio/high_mean": 0.0434071971103549, "clip_ratio/low_mean": 0.03159340703859925, "clip_ratio/low_min": 0.03159340703859925, "clip_ratio/region_mean": 0.07500060414895415, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 63.0, "completions/mean_terminated_length": 63.0, "completions/min_length": 58.0, "completions/min_terminated_length": 58.0, "entropy": 0.356488361954689, "epoch": 0.10824597341045106, "frac_reward_zero_std": 0.0, "grad_norm": 12.76325798034668, "learning_rate": 1.8363636363636365e-06, "loss": 0.064, "num_tokens": 6107356.0, "reward": 0.6803799867630005, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6803799867630005, "reward_meter_std": 0.3893775939941406, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3893775939941406, "reward_total_composite_mean": 0.6803799867630005, "reward_total_composite_std": 0.3893775939941406, "reward_total_mean": 0.6803799867630005, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6803799867630005, "rewards/meter/std": 0.3893775939941406, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6803799867630005, "rewards/total_composite/std": 0.3893775939941406, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008878231048584, "sampling/importance_sampling_ratio/min": 0.06511905789375305, "sampling/sampling_logp_difference/max": 2.7315380573272705, "sampling/sampling_logp_difference/mean": 0.09513663500547409, "step": 2695 }, { "clip_ratio/high_max": 0.016798642929643393, "clip_ratio/high_mean": 0.016798642929643393, "clip_ratio/low_mean": 0.01482684025540948, "clip_ratio/low_min": 0.01482684025540948, "clip_ratio/region_mean": 0.03162548318505287, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 66.625, "completions/mean_terminated_length": 66.625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.18041662964969873, "epoch": 0.10828613889223601, "frac_reward_zero_std": 0.0, "grad_norm": 9.303409576416016, "learning_rate": 1.8333333333333333e-06, "loss": 0.0176, "num_tokens": 6109361.0, "reward": 0.9331727027893066, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9331727027893066, "reward_meter_std": 0.008176974020898342, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008176976814866066, "reward_total_composite_mean": 0.9331727027893066, "reward_total_composite_std": 0.008176974020898342, "reward_total_mean": 0.9331727027893066, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9331727027893066, "rewards/meter/std": 0.008176974020898342, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9331727027893066, "rewards/total_composite/std": 0.008176974020898342, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9995071291923523, "sampling/importance_sampling_ratio/min": 0.3752399682998657, "sampling/sampling_logp_difference/max": 0.9801895618438721, "sampling/sampling_logp_difference/mean": 0.036672260612249374, "step": 2696 }, { "clip_ratio/high_max": 0.003729202609974891, "clip_ratio/high_mean": 0.003729202609974891, "clip_ratio/low_mean": 0.002214635896962136, "clip_ratio/low_min": 0.002214635896962136, "clip_ratio/region_mean": 0.005943838506937027, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 168.0, "completions/mean_terminated_length": 168.0, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.0484649995341897, "epoch": 0.10832630437402097, "frac_reward_zero_std": 0.0, "grad_norm": 0.7284796237945557, "learning_rate": 1.8303030303030305e-06, "loss": 0.0029, "num_tokens": 6112233.0, "reward": 0.9992363452911377, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992363452911377, "reward_meter_std": 9.161681373370811e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.160472109215334e-05, "reward_total_composite_mean": 0.9992363452911377, "reward_total_composite_std": 9.161681373370811e-05, "reward_total_mean": 0.9992363452911377, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992363452911377, "rewards/meter/std": 9.161681373370811e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992363452911377, "rewards/total_composite/std": 9.161681373370811e-05, "sampling/importance_sampling_ratio/max": 1.376589298248291, "sampling/importance_sampling_ratio/mean": 1.0003851652145386, "sampling/importance_sampling_ratio/min": 0.5037661194801331, "sampling/sampling_logp_difference/max": 0.685643196105957, "sampling/sampling_logp_difference/mean": 0.007861536927521229, "step": 2697 }, { "clip_ratio/high_max": 0.00870500784367323, "clip_ratio/high_mean": 0.00870500784367323, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.012276436435058713, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 71.625, "completions/mean_terminated_length": 71.625, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.051547068171203136, "epoch": 0.10836646985580592, "frac_reward_zero_std": 0.0, "grad_norm": 1.4812349081039429, "learning_rate": 1.8272727272727276e-06, "loss": -0.0019, "num_tokens": 6114086.0, "reward": 0.999380350112915, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999380350112915, "reward_meter_std": 0.00012280534429010004, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001228114851983264, "reward_total_composite_mean": 0.999380350112915, "reward_total_composite_std": 0.00012280534429010004, "reward_total_mean": 0.999380350112915, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999380350112915, "rewards/meter/std": 0.00012280534429010004, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999380350112915, "rewards/total_composite/std": 0.00012280534429010004, "sampling/importance_sampling_ratio/max": 1.2945717573165894, "sampling/importance_sampling_ratio/mean": 0.9971534013748169, "sampling/importance_sampling_ratio/min": 0.29206937551498413, "sampling/sampling_logp_difference/max": 1.2307639122009277, "sampling/sampling_logp_difference/mean": 0.012132828123867512, "step": 2698 }, { "clip_ratio/high_max": 0.007352941203862429, "clip_ratio/high_mean": 0.007352941203862429, "clip_ratio/low_mean": 0.01460084063000977, "clip_ratio/low_min": 0.01460084063000977, "clip_ratio/region_mean": 0.0219537818338722, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 34.125, "completions/mean_terminated_length": 34.125, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.18179897591471672, "epoch": 0.10840663533759087, "frac_reward_zero_std": 0.0, "grad_norm": 21.836759567260742, "learning_rate": 1.8242424242424244e-06, "loss": 0.0187, "num_tokens": 6115559.0, "reward": 0.9915792942047119, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9915792942047119, "reward_meter_std": 0.0012008182238787413, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001200812985189259, "reward_total_composite_mean": 0.9915792942047119, "reward_total_composite_std": 0.0012008182238787413, "reward_total_mean": 0.9915792942047119, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9915792942047119, "rewards/meter/std": 0.0012008182238787413, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9915792942047119, "rewards/total_composite/std": 0.0012008182238787413, "sampling/importance_sampling_ratio/max": 1.8209221363067627, "sampling/importance_sampling_ratio/mean": 1.0031808614730835, "sampling/importance_sampling_ratio/min": 0.3298044800758362, "sampling/sampling_logp_difference/max": 1.109255313873291, "sampling/sampling_logp_difference/mean": 0.024363644421100616, "step": 2699 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.625, "completions/mean_terminated_length": 71.625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.03858046652749181, "epoch": 0.10844680081937583, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.8212121212121215e-06, "loss": 0.0, "num_tokens": 6117340.0, "reward": 0.99944007396698, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99944007396698, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.99944007396698, "reward_total_composite_std": 0.0, "reward_total_mean": 0.99944007396698, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99944007396698, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99944007396698, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.2885301113128662, "sampling/importance_sampling_ratio/mean": 1.002069115638733, "sampling/importance_sampling_ratio/min": 0.7516875863075256, "sampling/sampling_logp_difference/max": 0.2854344844818115, "sampling/sampling_logp_difference/mean": 0.004740848205983639, "step": 2700 }, { "epoch": 0.10844680081937583, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 402.0, "eval_completions/max_terminated_length": 402.0, "eval_completions/mean_length": 205.91346153846155, "eval_completions/mean_terminated_length": 205.91346153846155, "eval_completions/min_length": 61.07692307692308, "eval_completions/min_terminated_length": 61.07692307692308, "eval_entropy": 0.2846654344063539, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 6117340.0, "eval_reward": 0.7056683852122381, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.9412473761118375, "eval_reward_count_adherence_std": 0.08066883425299938, "eval_reward_meter_mean": 0.818766135435838, "eval_reward_meter_std": 0.28128915272939664, "eval_reward_repeat_penalty_mean": 0.917067867058974, "eval_reward_repeat_penalty_std": 0.09873221929256733, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7056683852122381, "eval_reward_total_composite_std": 0.3022650325527558, "eval_reward_total_mean": 0.7056683852122381, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.9412473761118375, "eval_rewards/count_adherence/std": 0.08066883425299938, "eval_rewards/meter/mean": 0.818766135435838, "eval_rewards/meter/std": 0.28128915272939664, "eval_rewards/repeat_penalty/mean": 0.917067867058974, "eval_rewards/repeat_penalty/std": 0.09873221929256733, "eval_rewards/total_composite/mean": 0.7056683852122381, "eval_rewards/total_composite/std": 0.3022650325527558, "eval_runtime": 74.8526, "eval_samples_per_second": 1.389, "eval_sampling/importance_sampling_ratio/max": 1.4687065803087676, "eval_sampling/importance_sampling_ratio/mean": 1.00816364471729, "eval_sampling/importance_sampling_ratio/min": 0.36844213765401107, "eval_sampling/sampling_logp_difference/max": 1.0165979678814228, "eval_sampling/sampling_logp_difference/mean": 0.026865268770891886, "eval_steps_per_second": 0.174, "step": 2700 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.012467616703361273, "epoch": 0.10848696630116078, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.8181818181818183e-06, "loss": 0.0, "num_tokens": 6119068.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0299800634384155, "sampling/importance_sampling_ratio/mean": 0.9999621510505676, "sampling/importance_sampling_ratio/min": 0.7778228521347046, "sampling/sampling_logp_difference/max": 0.2512565851211548, "sampling/sampling_logp_difference/mean": 0.0014518008101731539, "step": 2701 }, { "clip_ratio/high_max": 0.007548359571956098, "clip_ratio/high_mean": 0.007548359571956098, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.00941403117030859, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 67.5, "completions/mean_terminated_length": 67.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.19097852148115635, "epoch": 0.10852713178294573, "frac_reward_zero_std": 0.0, "grad_norm": 3.3459887504577637, "learning_rate": 1.8151515151515153e-06, "loss": -0.001, "num_tokens": 6120768.0, "reward": 0.937604546546936, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.937604546546936, "reward_meter_std": 0.136740043759346, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.1367400586605072, "reward_total_composite_mean": 0.937604546546936, "reward_total_composite_std": 0.136740043759346, "reward_total_mean": 0.937604546546936, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.937604546546936, "rewards/meter/std": 0.136740043759346, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.937604546546936, "rewards/total_composite/std": 0.136740043759346, "sampling/importance_sampling_ratio/max": 1.6750621795654297, "sampling/importance_sampling_ratio/mean": 1.0090194940567017, "sampling/importance_sampling_ratio/min": 0.22187186777591705, "sampling/sampling_logp_difference/max": 1.505655288696289, "sampling/sampling_logp_difference/mean": 0.025002291426062584, "step": 2702 }, { "clip_ratio/high_max": 0.019398832926526666, "clip_ratio/high_mean": 0.019398832926526666, "clip_ratio/low_mean": 0.016337704844772816, "clip_ratio/low_min": 0.016337704844772816, "clip_ratio/region_mean": 0.03573653777129948, "completions/clipped_ratio": 0.0, "completions/max_length": 218.0, "completions/max_terminated_length": 218.0, "completions/mean_length": 213.75, "completions/mean_terminated_length": 213.75, "completions/min_length": 207.0, "completions/min_terminated_length": 207.0, "entropy": 0.220721784979105, "epoch": 0.10856729726473069, "frac_reward_zero_std": 0.0, "grad_norm": 2.7519359588623047, "learning_rate": 1.8121212121212124e-06, "loss": 0.0104, "num_tokens": 6124022.0, "reward": 0.9688572883605957, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976348876953125, "reward_meter_std": 0.0003981303598266095, "reward_repeat_penalty_mean": 0.9711538553237915, "reward_repeat_penalty_std": 0.039811473339796066, "reward_std": 0.03972842916846275, "reward_total_composite_mean": 0.9688572883605957, "reward_total_composite_std": 0.039728421717882156, "reward_total_mean": 0.9688572883605957, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976348876953125, "rewards/meter/std": 0.0003981303598266095, "rewards/repeat_penalty/mean": 0.9711538553237915, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.9688572883605957, "rewards/total_composite/std": 0.039728421717882156, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0066944360733032, "sampling/importance_sampling_ratio/min": 0.276298850774765, "sampling/sampling_logp_difference/max": 1.2862721681594849, "sampling/sampling_logp_difference/mean": 0.034267134964466095, "step": 2703 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.003703906899318099, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.007660032075364143, "epoch": 0.10860746274651564, "frac_reward_zero_std": 0.0, "grad_norm": 0.004991916939616203, "learning_rate": 1.8090909090909092e-06, "loss": -0.0003, "num_tokens": 6125799.0, "reward": 0.9981493353843689, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981493353843689, "reward_meter_std": 8.239712769864127e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.248746780736838e-06, "reward_total_composite_mean": 0.9981493353843689, "reward_total_composite_std": 8.239712769864127e-06, "reward_total_mean": 0.9981493353843689, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981493353843689, "rewards/meter/std": 8.239712769864127e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981493353843689, "rewards/total_composite/std": 8.239712769864127e-06, "sampling/importance_sampling_ratio/max": 1.0250744819641113, "sampling/importance_sampling_ratio/mean": 0.9988438487052917, "sampling/importance_sampling_ratio/min": 0.44041258096694946, "sampling/sampling_logp_difference/max": 0.8200433254241943, "sampling/sampling_logp_difference/mean": 0.0028709208127111197, "step": 2704 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0020572251814883202, "epoch": 0.1086476282283006, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.8060606060606063e-06, "loss": 0.0, "num_tokens": 6127407.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.002421498298645, "sampling/importance_sampling_ratio/mean": 1.0002423524856567, "sampling/importance_sampling_ratio/min": 0.9996000528335571, "sampling/sampling_logp_difference/max": 0.002418494550511241, "sampling/sampling_logp_difference/mean": 0.00024437991669401526, "step": 2705 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.002913154661655426, "epoch": 0.10868779371008555, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.803030303030303e-06, "loss": 0.0, "num_tokens": 6128951.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0118145942687988, "sampling/importance_sampling_ratio/mean": 1.0001347064971924, "sampling/importance_sampling_ratio/min": 0.9924525022506714, "sampling/sampling_logp_difference/max": 0.011745302006602287, "sampling/sampling_logp_difference/mean": 0.0002643416519276798, "step": 2706 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.001665496762143448, "epoch": 0.1087279591918705, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.8000000000000001e-06, "loss": 0.0, "num_tokens": 6130687.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0022114515304565, "sampling/importance_sampling_ratio/mean": 1.0002092123031616, "sampling/importance_sampling_ratio/min": 0.9998108148574829, "sampling/sampling_logp_difference/max": 0.0022089765407145023, "sampling/sampling_logp_difference/mean": 0.00021008482144679874, "step": 2707 }, { "clip_ratio/high_max": 0.004673305433243513, "clip_ratio/high_mean": 0.004673305433243513, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.004673305433243513, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 106.5, "completions/mean_terminated_length": 106.5, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.053598769940435886, "epoch": 0.10876812467365546, "frac_reward_zero_std": 0.0, "grad_norm": 1.1375532150268555, "learning_rate": 1.796969696969697e-06, "loss": -0.0031, "num_tokens": 6132899.0, "reward": 0.999268651008606, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999268651008606, "reward_meter_std": 0.00017192552331835032, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00017192917584907264, "reward_total_composite_mean": 0.999268651008606, "reward_total_composite_std": 0.00017192552331835032, "reward_total_mean": 0.999268651008606, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999268651008606, "rewards/meter/std": 0.00017192552331835032, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999268651008606, "rewards/total_composite/std": 0.00017192552331835032, "sampling/importance_sampling_ratio/max": 1.3604763746261597, "sampling/importance_sampling_ratio/mean": 1.0018383264541626, "sampling/importance_sampling_ratio/min": 0.44022059440612793, "sampling/sampling_logp_difference/max": 0.8204793930053711, "sampling/sampling_logp_difference/mean": 0.008916010148823261, "step": 2708 }, { "clip_ratio/high_max": 0.01722477504517883, "clip_ratio/high_mean": 0.01722477504517883, "clip_ratio/low_mean": 0.010879121022298932, "clip_ratio/low_min": 0.010879121022298932, "clip_ratio/region_mean": 0.028103896067477763, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 102.5, "completions/mean_terminated_length": 102.5, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.2976728491485119, "epoch": 0.10880829015544041, "frac_reward_zero_std": 0.0, "grad_norm": 3.0748214721679688, "learning_rate": 1.793939393939394e-06, "loss": 0.0051, "num_tokens": 6135215.0, "reward": 0.998659610748291, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998659610748291, "reward_meter_std": 0.0006201548385433853, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006201544310897589, "reward_total_composite_mean": 0.998659610748291, "reward_total_composite_std": 0.0006201548385433853, "reward_total_mean": 0.998659610748291, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998659610748291, "rewards/meter/std": 0.0006201548385433853, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998659610748291, "rewards/total_composite/std": 0.0006201548385433853, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0111318826675415, "sampling/importance_sampling_ratio/min": 0.35953280329704285, "sampling/sampling_logp_difference/max": 1.1598763465881348, "sampling/sampling_logp_difference/mean": 0.03828549385070801, "step": 2709 }, { "clip_ratio/high_max": 0.001754407538101077, "clip_ratio/high_mean": 0.001754407538101077, "clip_ratio/low_mean": 0.003502913983538747, "clip_ratio/low_min": 0.003502913983538747, "clip_ratio/region_mean": 0.005257321521639824, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 142.5, "completions/mean_terminated_length": 142.5, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 0.0695994533598423, "epoch": 0.10884845563722537, "frac_reward_zero_std": 0.0, "grad_norm": 1.891419529914856, "learning_rate": 1.7909090909090908e-06, "loss": 0.0027, "num_tokens": 6137547.0, "reward": 0.8386699557304382, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999262809753418, "reward_meter_std": 0.00011097361129941419, "reward_repeat_penalty_mean": 0.8392857313156128, "reward_repeat_penalty_std": 0.09155284613370895, "reward_std": 0.0915173590183258, "reward_total_composite_mean": 0.8386699557304382, "reward_total_composite_std": 0.0915173664689064, "reward_total_mean": 0.8386699557304382, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999262809753418, "rewards/meter/std": 0.00011097361129941419, "rewards/repeat_penalty/mean": 0.8392857313156128, "rewards/repeat_penalty/std": 0.09155284613370895, "rewards/total_composite/mean": 0.8386699557304382, "rewards/total_composite/std": 0.0915173664689064, "sampling/importance_sampling_ratio/max": 1.472945213317871, "sampling/importance_sampling_ratio/mean": 1.0032413005828857, "sampling/importance_sampling_ratio/min": 0.5134629607200623, "sampling/sampling_logp_difference/max": 0.6665773391723633, "sampling/sampling_logp_difference/mean": 0.010055704973638058, "step": 2710 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.25, "completions/mean_terminated_length": 66.25, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.016254089423455298, "epoch": 0.10888862111901032, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.787878787878788e-06, "loss": 0.0, "num_tokens": 6139309.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.395367980003357, "sampling/importance_sampling_ratio/mean": 1.0023350715637207, "sampling/importance_sampling_ratio/min": 0.729793906211853, "sampling/sampling_logp_difference/max": 0.3331582546234131, "sampling/sampling_logp_difference/mean": 0.0032224601600319147, "step": 2711 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.0020491802133619785, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 60.75, "completions/mean_terminated_length": 60.75, "completions/min_length": 59.0, "completions/min_terminated_length": 59.0, "entropy": 0.02895409008488059, "epoch": 0.10892878660079527, "frac_reward_zero_std": 0.0, "grad_norm": 3.4030656814575195, "learning_rate": 1.7848484848484851e-06, "loss": -0.0068, "num_tokens": 6141147.0, "reward": 0.9972513318061829, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972513318061829, "reward_meter_std": 0.0001800779573386535, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00018008207553066313, "reward_total_composite_mean": 0.9972513318061829, "reward_total_composite_std": 0.0001800779573386535, "reward_total_mean": 0.9972513318061829, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972513318061829, "rewards/meter/std": 0.0001800779573386535, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972513318061829, "rewards/total_composite/std": 0.0001800779573386535, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0053867101669312, "sampling/importance_sampling_ratio/min": 0.7316676378250122, "sampling/sampling_logp_difference/max": 0.9685866832733154, "sampling/sampling_logp_difference/mean": 0.0070107607170939445, "step": 2712 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.004417360032675788, "epoch": 0.10896895208258023, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.781818181818182e-06, "loss": 0.0, "num_tokens": 6142939.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.023779034614563, "sampling/importance_sampling_ratio/mean": 0.9994007349014282, "sampling/importance_sampling_ratio/min": 0.7536472678184509, "sampling/sampling_logp_difference/max": 0.2828308939933777, "sampling/sampling_logp_difference/mean": 0.0010938982013612986, "step": 2713 }, { "clip_ratio/high_max": 0.016068846685811877, "clip_ratio/high_mean": 0.016068846685811877, "clip_ratio/low_mean": 0.0015432098880410194, "clip_ratio/low_min": 0.0015432098880410194, "clip_ratio/region_mean": 0.017612056573852897, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 78.875, "completions/mean_terminated_length": 78.875, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.15530052781105042, "epoch": 0.10900911756436518, "frac_reward_zero_std": 0.0, "grad_norm": 2.3612427711486816, "learning_rate": 1.778787878787879e-06, "loss": 0.0083, "num_tokens": 6144770.0, "reward": 0.9988387227058411, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988387227058411, "reward_meter_std": 0.0002311533025931567, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00023116436204873025, "reward_total_composite_mean": 0.9988387227058411, "reward_total_composite_std": 0.0002311533025931567, "reward_total_mean": 0.9988387227058411, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988387227058411, "rewards/meter/std": 0.0002311533025931567, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988387227058411, "rewards/total_composite/std": 0.0002311533025931567, "sampling/importance_sampling_ratio/max": 1.6622776985168457, "sampling/importance_sampling_ratio/mean": 1.0010093450546265, "sampling/importance_sampling_ratio/min": 0.3075661063194275, "sampling/sampling_logp_difference/max": 1.179065227508545, "sampling/sampling_logp_difference/mean": 0.028173334896564484, "step": 2714 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0015674135065637529, "clip_ratio/low_min": 0.0015674135065637529, "clip_ratio/region_mean": 0.0015674135065637529, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 159.25, "completions/mean_terminated_length": 159.25, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.08590239379554987, "epoch": 0.10904928304615014, "frac_reward_zero_std": 0.0, "grad_norm": 0.7956473231315613, "learning_rate": 1.775757575757576e-06, "loss": 0.0025, "num_tokens": 6147492.0, "reward": 0.9009125232696533, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979335069656372, "reward_meter_std": 1.7607557310839184e-05, "reward_repeat_penalty_mean": 0.9027777910232544, "reward_repeat_penalty_std": 0.03928370773792267, "reward_std": 0.03921113163232803, "reward_total_composite_mean": 0.9009125232696533, "reward_total_composite_std": 0.03921113535761833, "reward_total_mean": 0.9009125232696533, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979335069656372, "rewards/meter/std": 1.7607557310839184e-05, "rewards/repeat_penalty/mean": 0.9027777910232544, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.9009125232696533, "rewards/total_composite/std": 0.03921113535761833, "sampling/importance_sampling_ratio/max": 1.4207649230957031, "sampling/importance_sampling_ratio/mean": 1.0039114952087402, "sampling/importance_sampling_ratio/min": 0.5508456826210022, "sampling/sampling_logp_difference/max": 0.5963006019592285, "sampling/sampling_logp_difference/mean": 0.009351929649710655, "step": 2715 }, { "clip_ratio/high_max": 0.014018616639077663, "clip_ratio/high_mean": 0.014018616639077663, "clip_ratio/low_mean": 0.008343940833583474, "clip_ratio/low_min": 0.008343940833583474, "clip_ratio/region_mean": 0.022362557472661138, "completions/clipped_ratio": 0.0, "completions/max_length": 459.0, "completions/max_terminated_length": 459.0, "completions/mean_length": 438.125, "completions/mean_terminated_length": 438.125, "completions/min_length": 361.0, "completions/min_terminated_length": 361.0, "entropy": 0.3533891662955284, "epoch": 0.10908944852793509, "frac_reward_zero_std": 0.0, "grad_norm": 1.8313432931900024, "learning_rate": 1.7727272727272729e-06, "loss": -0.0565, "num_tokens": 6152509.0, "reward": 0.7314679622650146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7666666507720947, "reward_count_adherence_std": 0.07126966118812561, "reward_meter_mean": 0.9989956617355347, "reward_meter_std": 0.00016425758076366037, "reward_repeat_penalty_mean": 0.9524672031402588, "reward_repeat_penalty_std": 0.05135857313871384, "reward_std": 0.09445606172084808, "reward_total_composite_mean": 0.7314679622650146, "reward_total_composite_std": 0.09445607662200928, "reward_total_mean": 0.7314679622650146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7666666507720947, "rewards/count_adherence/std": 0.07126966118812561, "rewards/meter/mean": 0.9989956617355347, "rewards/meter/std": 0.00016425758076366037, "rewards/repeat_penalty/mean": 0.9524672031402588, "rewards/repeat_penalty/std": 0.05135857313871384, "rewards/total_composite/mean": 0.7314679622650146, "rewards/total_composite/std": 0.09445607662200928, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0078824758529663, "sampling/importance_sampling_ratio/min": 0.15290868282318115, "sampling/sampling_logp_difference/max": 1.8779144287109375, "sampling/sampling_logp_difference/mean": 0.04289248585700989, "step": 2716 }, { "clip_ratio/high_max": 0.00957868475234136, "clip_ratio/high_mean": 0.00957868475234136, "clip_ratio/low_mean": 0.0056604581186547875, "clip_ratio/low_min": 0.0056604581186547875, "clip_ratio/region_mean": 0.015239142870996147, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 131.375, "completions/mean_terminated_length": 131.375, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.05451698787510395, "epoch": 0.10912961400972004, "frac_reward_zero_std": 0.0, "grad_norm": 1.0715020895004272, "learning_rate": 1.76969696969697e-06, "loss": 0.0033, "num_tokens": 6155040.0, "reward": 0.9993519186973572, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993519186973572, "reward_meter_std": 4.195739529677667e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.196614827378653e-05, "reward_total_composite_mean": 0.9993519186973572, "reward_total_composite_std": 4.195739529677667e-05, "reward_total_mean": 0.9993519186973572, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993519186973572, "rewards/meter/std": 4.195739529677667e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993519186973572, "rewards/total_composite/std": 4.195739529677667e-05, "sampling/importance_sampling_ratio/max": 1.3561691045761108, "sampling/importance_sampling_ratio/mean": 0.9986187815666199, "sampling/importance_sampling_ratio/min": 0.35653650760650635, "sampling/sampling_logp_difference/max": 1.0313186645507812, "sampling/sampling_logp_difference/mean": 0.011481878347694874, "step": 2717 }, { "clip_ratio/high_max": 0.027308928780257702, "clip_ratio/high_mean": 0.027308928780257702, "clip_ratio/low_mean": 0.023148135747760534, "clip_ratio/low_min": 0.023148135747760534, "clip_ratio/region_mean": 0.050457064528018236, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 307.25, "completions/mean_terminated_length": 307.25, "completions/min_length": 298.0, "completions/min_terminated_length": 298.0, "entropy": 0.5477991737425327, "epoch": 0.109169779491505, "frac_reward_zero_std": 0.0, "grad_norm": 3.4733986854553223, "learning_rate": 1.7666666666666668e-06, "loss": 0.0052, "num_tokens": 6159098.0, "reward": 0.8657643795013428, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986767768859863, "reward_meter_std": 0.0004087568959221244, "reward_repeat_penalty_mean": 0.9632352590560913, "reward_repeat_penalty_std": 0.04376610368490219, "reward_std": 0.03933200240135193, "reward_total_composite_mean": 0.8657643795013428, "reward_total_composite_std": 0.039332009851932526, "reward_total_mean": 0.8657643795013428, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986767768859863, "rewards/meter/std": 0.0004087568959221244, "rewards/repeat_penalty/mean": 0.9632352590560913, "rewards/repeat_penalty/std": 0.04376610368490219, "rewards/total_composite/mean": 0.8657643795013428, "rewards/total_composite/std": 0.039332009851932526, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010833740234375, "sampling/importance_sampling_ratio/min": 0.02132892794907093, "sampling/sampling_logp_difference/max": 3.847691059112549, "sampling/sampling_logp_difference/mean": 0.059995777904987335, "step": 2718 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.00040178035851567984, "epoch": 0.10920994497328995, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.7636363636363638e-06, "loss": 0.0, "num_tokens": 6160514.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.000882625579834, "sampling/importance_sampling_ratio/mean": 1.0000470876693726, "sampling/importance_sampling_ratio/min": 0.9999879002571106, "sampling/sampling_logp_difference/max": 0.0008822012459859252, "sampling/sampling_logp_difference/mean": 4.7315745177911595e-05, "step": 2719 }, { "clip_ratio/high_max": 0.005076272878795862, "clip_ratio/high_mean": 0.005076272878795862, "clip_ratio/low_mean": 0.0024999999441206455, "clip_ratio/low_min": 0.0024999999441206455, "clip_ratio/region_mean": 0.007576272822916508, "completions/clipped_ratio": 0.0, "completions/max_length": 100.0, "completions/max_terminated_length": 100.0, "completions/mean_length": 98.5, "completions/mean_terminated_length": 98.5, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.03585473005659878, "epoch": 0.1092501104550749, "frac_reward_zero_std": 0.0, "grad_norm": 0.7155827283859253, "learning_rate": 1.7606060606060606e-06, "loss": 0.0004, "num_tokens": 6162606.0, "reward": 0.9994127750396729, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994127750396729, "reward_meter_std": 2.115286042680964e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.114395101671107e-05, "reward_total_composite_mean": 0.9994127750396729, "reward_total_composite_std": 2.115286042680964e-05, "reward_total_mean": 0.9994127750396729, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994127750396729, "rewards/meter/std": 2.115286042680964e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994127750396729, "rewards/total_composite/std": 2.115286042680964e-05, "sampling/importance_sampling_ratio/max": 1.7030972242355347, "sampling/importance_sampling_ratio/mean": 1.0007652044296265, "sampling/importance_sampling_ratio/min": 0.3942791223526001, "sampling/sampling_logp_difference/max": 0.9306962490081787, "sampling/sampling_logp_difference/mean": 0.008277514949440956, "step": 2720 }, { "clip_ratio/high_max": 0.02390979148913175, "clip_ratio/high_mean": 0.02390979148913175, "clip_ratio/low_mean": 0.017159562092274427, "clip_ratio/low_min": 0.017159562092274427, "clip_ratio/region_mean": 0.041069353581406176, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 70.5, "completions/mean_terminated_length": 70.5, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.3494097013026476, "epoch": 0.10929027593685986, "frac_reward_zero_std": 0.0, "grad_norm": 5.66716194152832, "learning_rate": 1.7575757575757577e-06, "loss": 0.0456, "num_tokens": 6164482.0, "reward": 0.9914366006851196, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9914366006851196, "reward_meter_std": 0.004417366813868284, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004417361691594124, "reward_total_composite_mean": 0.9914366006851196, "reward_total_composite_std": 0.004417366813868284, "reward_total_mean": 0.9914366006851196, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9914366006851196, "rewards/meter/std": 0.004417366813868284, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9914366006851196, "rewards/total_composite/std": 0.004417366813868284, "sampling/importance_sampling_ratio/max": 1.5672898292541504, "sampling/importance_sampling_ratio/mean": 0.998794436454773, "sampling/importance_sampling_ratio/min": 0.2594565749168396, "sampling/sampling_logp_difference/max": 1.349165916442871, "sampling/sampling_logp_difference/mean": 0.054880011826753616, "step": 2721 }, { "clip_ratio/high_max": 0.027233949513174593, "clip_ratio/high_mean": 0.027233949513174593, "clip_ratio/low_mean": 0.005281690042465925, "clip_ratio/low_min": 0.005281690042465925, "clip_ratio/region_mean": 0.03251563955564052, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.22193360701203346, "epoch": 0.10933044141864481, "frac_reward_zero_std": 0.0, "grad_norm": 11.017292976379395, "learning_rate": 1.7545454545454545e-06, "loss": 0.0176, "num_tokens": 6166347.0, "reward": 0.9949984550476074, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949984550476074, "reward_meter_std": 0.011957395821809769, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01195740420371294, "reward_total_composite_mean": 0.9949984550476074, "reward_total_composite_std": 0.011957395821809769, "reward_total_mean": 0.9949984550476074, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949984550476074, "rewards/meter/std": 0.011957395821809769, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949984550476074, "rewards/total_composite/std": 0.011957395821809769, "sampling/importance_sampling_ratio/max": 1.5986580848693848, "sampling/importance_sampling_ratio/mean": 0.996598482131958, "sampling/importance_sampling_ratio/min": 0.05794627591967583, "sampling/sampling_logp_difference/max": 2.848238945007324, "sampling/sampling_logp_difference/mean": 0.05068688467144966, "step": 2722 }, { "clip_ratio/high_max": 0.03575678775086999, "clip_ratio/high_mean": 0.03575678775086999, "clip_ratio/low_mean": 0.013798366067931056, "clip_ratio/low_min": 0.013798366067931056, "clip_ratio/region_mean": 0.049555153818801045, "completions/clipped_ratio": 0.0, "completions/max_length": 242.0, "completions/max_terminated_length": 242.0, "completions/mean_length": 235.25, "completions/mean_terminated_length": 235.25, "completions/min_length": 225.0, "completions/min_terminated_length": 225.0, "entropy": 0.5150004588067532, "epoch": 0.10937060690042977, "frac_reward_zero_std": 0.0, "grad_norm": 2.760967254638672, "learning_rate": 1.7515151515151516e-06, "loss": 0.0093, "num_tokens": 6169685.0, "reward": 0.969149649143219, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979515671730042, "reward_meter_std": 0.0015659787459298968, "reward_repeat_penalty_mean": 0.9711538553237915, "reward_repeat_penalty_std": 0.039811473339796066, "reward_std": 0.0393538735806942, "reward_total_composite_mean": 0.969149649143219, "reward_total_composite_std": 0.0393538773059845, "reward_total_mean": 0.969149649143219, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979515671730042, "rewards/meter/std": 0.0015659787459298968, "rewards/repeat_penalty/mean": 0.9711538553237915, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.969149649143219, "rewards/total_composite/std": 0.0393538773059845, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0089517831802368, "sampling/importance_sampling_ratio/min": 0.2813614308834076, "sampling/sampling_logp_difference/max": 1.2681152820587158, "sampling/sampling_logp_difference/mean": 0.05202290415763855, "step": 2723 }, { "clip_ratio/high_max": 0.0387622892158106, "clip_ratio/high_mean": 0.0387622892158106, "clip_ratio/low_mean": 0.014925372786819935, "clip_ratio/low_min": 0.014925372786819935, "clip_ratio/region_mean": 0.05368766200263053, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.125, "completions/mean_terminated_length": 65.125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.36046687327325344, "epoch": 0.10941077238221472, "frac_reward_zero_std": 0.0, "grad_norm": 13.60006046295166, "learning_rate": 1.7484848484848486e-06, "loss": 0.0275, "num_tokens": 6171486.0, "reward": 0.9050167798995972, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9050167798995972, "reward_meter_std": 0.11941422522068024, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.11941422522068024, "reward_total_composite_mean": 0.9050167798995972, "reward_total_composite_std": 0.11941422522068024, "reward_total_mean": 0.9050167798995972, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9050167798995972, "rewards/meter/std": 0.11941422522068024, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9050167798995972, "rewards/total_composite/std": 0.11941422522068024, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9989842772483826, "sampling/importance_sampling_ratio/min": 0.14376457035541534, "sampling/sampling_logp_difference/max": 1.9395782947540283, "sampling/sampling_logp_difference/mean": 0.07930425554513931, "step": 2724 }, { "clip_ratio/high_max": 0.012341346475295722, "clip_ratio/high_mean": 0.012341346475295722, "clip_ratio/low_mean": 0.01608805637806654, "clip_ratio/low_min": 0.01608805637806654, "clip_ratio/region_mean": 0.028429402853362262, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.25, "completions/mean_terminated_length": 70.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.2636737637221813, "epoch": 0.10945093786399968, "frac_reward_zero_std": 0.0, "grad_norm": 5.4172844886779785, "learning_rate": 1.7454545454545456e-06, "loss": -0.0057, "num_tokens": 6173408.0, "reward": 0.9948872923851013, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9948872923851013, "reward_meter_std": 0.0014726987574249506, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0014726977096870542, "reward_total_composite_mean": 0.9948872923851013, "reward_total_composite_std": 0.0014726987574249506, "reward_total_mean": 0.9948872923851013, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9948872923851013, "rewards/meter/std": 0.0014726987574249506, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9948872923851013, "rewards/total_composite/std": 0.0014726987574249506, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0125356912612915, "sampling/importance_sampling_ratio/min": 0.28223544359207153, "sampling/sampling_logp_difference/max": 1.2650136947631836, "sampling/sampling_logp_difference/mean": 0.04584960639476776, "step": 2725 }, { "clip_ratio/high_max": 0.009652552893385291, "clip_ratio/high_mean": 0.009652552893385291, "clip_ratio/low_mean": 0.01458781841211021, "clip_ratio/low_min": 0.01458781841211021, "clip_ratio/region_mean": 0.0242403713054955, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 103.875, "completions/mean_terminated_length": 103.875, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.3632195107638836, "epoch": 0.10949110334578463, "frac_reward_zero_std": 0.0, "grad_norm": 2.82613468170166, "learning_rate": 1.7424242424242427e-06, "loss": -0.004, "num_tokens": 6175647.0, "reward": 0.9729191064834595, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9729191064834595, "reward_meter_std": 0.010998339392244816, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010998345911502838, "reward_total_composite_mean": 0.9729191064834595, "reward_total_composite_std": 0.010998339392244816, "reward_total_mean": 0.9729191064834595, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9729191064834595, "rewards/meter/std": 0.010998339392244816, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9729191064834595, "rewards/total_composite/std": 0.010998339392244816, "sampling/importance_sampling_ratio/max": 1.7675621509552002, "sampling/importance_sampling_ratio/mean": 1.0128881931304932, "sampling/importance_sampling_ratio/min": 0.27118435502052307, "sampling/sampling_logp_difference/max": 1.3049564361572266, "sampling/sampling_logp_difference/mean": 0.0373404435813427, "step": 2726 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.125, "completions/mean_terminated_length": 71.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.03847737517207861, "epoch": 0.10953126882756958, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.7393939393939397e-06, "loss": 0.0, "num_tokens": 6177392.0, "reward": 0.99944007396698, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99944007396698, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.99944007396698, "reward_total_composite_std": 0.0, "reward_total_mean": 0.99944007396698, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99944007396698, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99944007396698, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0543627738952637, "sampling/importance_sampling_ratio/mean": 1.0021454095840454, "sampling/importance_sampling_ratio/min": 0.7279666066169739, "sampling/sampling_logp_difference/max": 0.31750011444091797, "sampling/sampling_logp_difference/mean": 0.004035182762891054, "step": 2727 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.011132915853522718, "epoch": 0.10957143430935454, "frac_reward_zero_std": 0.0, "grad_norm": 1.235407829284668, "learning_rate": 1.7363636363636366e-06, "loss": -0.0004, "num_tokens": 6178952.0, "reward": 0.9992912411689758, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992912411689758, "reward_meter_std": 1.968258584383875e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.9688604879775085e-05, "reward_total_composite_mean": 0.9992912411689758, "reward_total_composite_std": 1.968258584383875e-05, "reward_total_mean": 0.9992912411689758, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992912411689758, "rewards/meter/std": 1.968258584383875e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992912411689758, "rewards/total_composite/std": 1.968258584383875e-05, "sampling/importance_sampling_ratio/max": 1.0773866176605225, "sampling/importance_sampling_ratio/mean": 0.9995523691177368, "sampling/importance_sampling_ratio/min": 0.6159699559211731, "sampling/sampling_logp_difference/max": 0.4845571517944336, "sampling/sampling_logp_difference/mean": 0.0030222726054489613, "step": 2728 }, { "clip_ratio/high_max": 0.0216883085668087, "clip_ratio/high_mean": 0.0216883085668087, "clip_ratio/low_mean": 0.01417952380143106, "clip_ratio/low_min": 0.01417952380143106, "clip_ratio/region_mean": 0.03586783236823976, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 345.0, "completions/mean_terminated_length": 345.0, "completions/min_length": 339.0, "completions/min_terminated_length": 339.0, "entropy": 0.4432450830936432, "epoch": 0.10961159979113949, "frac_reward_zero_std": 0.0, "grad_norm": 1.5231090784072876, "learning_rate": 1.7333333333333336e-06, "loss": 0.0017, "num_tokens": 6183584.0, "reward": 0.8594046235084534, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989666938781738, "reward_meter_std": 0.00017414879403077066, "reward_repeat_penalty_mean": 0.9558823108673096, "reward_repeat_penalty_std": 0.0608881339430809, "reward_std": 0.05473008751869202, "reward_total_composite_mean": 0.8594046235084534, "reward_total_composite_std": 0.05473008379340172, "reward_total_mean": 0.8594046235084534, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989666938781738, "rewards/meter/std": 0.00017414879403077066, "rewards/repeat_penalty/mean": 0.9558823108673096, "rewards/repeat_penalty/std": 0.0608881339430809, "rewards/total_composite/mean": 0.8594046235084534, "rewards/total_composite/std": 0.05473008379340172, "sampling/importance_sampling_ratio/max": 1.6215828657150269, "sampling/importance_sampling_ratio/mean": 1.0122843980789185, "sampling/importance_sampling_ratio/min": 0.2652393579483032, "sampling/sampling_logp_difference/max": 1.327122688293457, "sampling/sampling_logp_difference/mean": 0.05097750574350357, "step": 2729 }, { "clip_ratio/high_max": 0.061030346201732755, "clip_ratio/high_mean": 0.061030346201732755, "clip_ratio/low_mean": 0.017045455053448677, "clip_ratio/low_min": 0.017045455053448677, "clip_ratio/region_mean": 0.07807580125518143, "completions/clipped_ratio": 0.0, "completions/max_length": 48.0, "completions/max_terminated_length": 48.0, "completions/mean_length": 44.625, "completions/mean_terminated_length": 44.625, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.37994090653955936, "epoch": 0.10965176527292445, "frac_reward_zero_std": 0.0, "grad_norm": 14.159218788146973, "learning_rate": 1.7303030303030304e-06, "loss": -0.0005, "num_tokens": 6185301.0, "reward": 0.9303584098815918, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9303584098815918, "reward_meter_std": 0.03593900799751282, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.035938993096351624, "reward_total_composite_mean": 0.9303584098815918, "reward_total_composite_std": 0.03593900799751282, "reward_total_mean": 0.9303584098815918, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9303584098815918, "rewards/meter/std": 0.03593900799751282, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9303584098815918, "rewards/total_composite/std": 0.03593900799751282, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.024146556854248, "sampling/importance_sampling_ratio/min": 0.0019010662799701095, "sampling/sampling_logp_difference/max": 6.265340328216553, "sampling/sampling_logp_difference/mean": 0.1046677976846695, "step": 2730 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.003038628783542663, "epoch": 0.1096919307547094, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.7272727272727275e-06, "loss": 0.0, "num_tokens": 6187213.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0230506658554077, "sampling/importance_sampling_ratio/mean": 0.9999112486839294, "sampling/importance_sampling_ratio/min": 0.9334831237792969, "sampling/sampling_logp_difference/max": 0.0688324123620987, "sampling/sampling_logp_difference/mean": 0.0005151918157935143, "step": 2731 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.0015087932988535613, "epoch": 0.10973209623649435, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.7242424242424243e-06, "loss": 0.0, "num_tokens": 6188637.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0033706426620483, "sampling/importance_sampling_ratio/mean": 1.0000722408294678, "sampling/importance_sampling_ratio/min": 0.9961644411087036, "sampling/sampling_logp_difference/max": 0.003842935897409916, "sampling/sampling_logp_difference/mean": 0.0001443149521946907, "step": 2732 }, { "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/region_mean": 0.0070436508394777775, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.75, "completions/mean_terminated_length": 34.75, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.12251334544271231, "epoch": 0.10977226171827931, "frac_reward_zero_std": 0.0, "grad_norm": 22.582195281982422, "learning_rate": 1.7212121212121214e-06, "loss": 0.0137, "num_tokens": 6190107.0, "reward": 0.945763885974884, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.945763885974884, "reward_meter_std": 0.12533576786518097, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.12533576786518097, "reward_total_composite_mean": 0.945763885974884, "reward_total_composite_std": 0.12533576786518097, "reward_total_mean": 0.945763885974884, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.945763885974884, "rewards/meter/std": 0.12533576786518097, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.945763885974884, "rewards/total_composite/std": 0.12533576786518097, "sampling/importance_sampling_ratio/max": 1.4781155586242676, "sampling/importance_sampling_ratio/mean": 1.0029886960983276, "sampling/importance_sampling_ratio/min": 0.2279849797487259, "sampling/sampling_logp_difference/max": 1.478475570678711, "sampling/sampling_logp_difference/mean": 0.02166467346251011, "step": 2733 }, { "clip_ratio/high_max": 0.033498246455565095, "clip_ratio/high_mean": 0.033498246455565095, "clip_ratio/low_mean": 0.004182156175374985, "clip_ratio/low_min": 0.004182156175374985, "clip_ratio/region_mean": 0.03768040263094008, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 268.875, "completions/mean_terminated_length": 268.875, "completions/min_length": 255.0, "completions/min_terminated_length": 255.0, "entropy": 0.43885183334350586, "epoch": 0.10981242720006426, "frac_reward_zero_std": 0.0, "grad_norm": 2.762172222137451, "learning_rate": 1.7181818181818182e-06, "loss": 0.0021, "num_tokens": 6193858.0, "reward": 0.9894447326660156, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990511536598206, "reward_meter_std": 0.00018971240206155926, "reward_repeat_penalty_mean": 0.990384578704834, "reward_repeat_penalty_std": 0.027196412906050682, "reward_std": 0.02716328203678131, "reward_total_composite_mean": 0.9894447326660156, "reward_total_composite_std": 0.027163289487361908, "reward_total_mean": 0.9894447326660156, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990511536598206, "rewards/meter/std": 0.00018971240206155926, "rewards/repeat_penalty/mean": 0.990384578704834, "rewards/repeat_penalty/std": 0.027196412906050682, "rewards/total_composite/mean": 0.9894447326660156, "rewards/total_composite/std": 0.027163289487361908, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0080488920211792, "sampling/importance_sampling_ratio/min": 0.12064424902200699, "sampling/sampling_logp_difference/max": 2.1149091720581055, "sampling/sampling_logp_difference/mean": 0.04823000356554985, "step": 2734 }, { "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/region_mean": 0.008878269698470831, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 70.875, "completions/mean_terminated_length": 70.875, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.05470029590651393, "epoch": 0.10985259268184921, "frac_reward_zero_std": 0.0, "grad_norm": 2.1266746520996094, "learning_rate": 1.7151515151515152e-06, "loss": -0.0023, "num_tokens": 6195785.0, "reward": 0.999404788017273, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999404788017273, "reward_meter_std": 0.00010114238102687523, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010115459008375183, "reward_total_composite_mean": 0.999404788017273, "reward_total_composite_std": 0.00010114238102687523, "reward_total_mean": 0.999404788017273, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999404788017273, "rewards/meter/std": 0.00010114238102687523, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999404788017273, "rewards/total_composite/std": 0.00010114238102687523, "sampling/importance_sampling_ratio/max": 1.3098119497299194, "sampling/importance_sampling_ratio/mean": 1.0012547969818115, "sampling/importance_sampling_ratio/min": 0.5381800532341003, "sampling/sampling_logp_difference/max": 0.6195621490478516, "sampling/sampling_logp_difference/mean": 0.007956513203680515, "step": 2735 }, { "clip_ratio/high_max": 0.023400326492264867, "clip_ratio/high_mean": 0.023400326492264867, "clip_ratio/low_mean": 0.004048783564940095, "clip_ratio/low_min": 0.004048783564940095, "clip_ratio/region_mean": 0.027449110057204962, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 122.625, "completions/mean_terminated_length": 122.625, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.1498414622619748, "epoch": 0.10989275816363417, "frac_reward_zero_std": 0.0, "grad_norm": 2.7460949420928955, "learning_rate": 1.7121212121212123e-06, "loss": 0.0073, "num_tokens": 6198070.0, "reward": 0.9442275762557983, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976873397827148, "reward_meter_std": 0.0002810863661579788, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07359235733747482, "reward_total_composite_mean": 0.9442275762557983, "reward_total_composite_std": 0.07359237223863602, "reward_total_mean": 0.9442275762557983, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976873397827148, "rewards/meter/std": 0.0002810863661579788, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9442275762557983, "rewards/total_composite/std": 0.07359237223863602, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9990551471710205, "sampling/importance_sampling_ratio/min": 0.4150184392929077, "sampling/sampling_logp_difference/max": 0.8794323205947876, "sampling/sampling_logp_difference/mean": 0.02476387470960617, "step": 2736 }, { "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0012755101779475808, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.875, "completions/mean_terminated_length": 97.875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.015126468031667173, "epoch": 0.10993292364541912, "frac_reward_zero_std": 0.0, "grad_norm": 0.5398823618888855, "learning_rate": 1.7090909090909091e-06, "loss": -0.001, "num_tokens": 6200301.0, "reward": 0.9994151592254639, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994151592254639, "reward_meter_std": 1.6079073247965425e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.6088057236629538e-05, "reward_total_composite_mean": 0.9994151592254639, "reward_total_composite_std": 1.6079073247965425e-05, "reward_total_mean": 0.9994151592254639, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994151592254639, "rewards/meter/std": 1.6079073247965425e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994151592254639, "rewards/total_composite/std": 1.6079073247965425e-05, "sampling/importance_sampling_ratio/max": 1.3984962701797485, "sampling/importance_sampling_ratio/mean": 0.9999629259109497, "sampling/importance_sampling_ratio/min": 0.16443268954753876, "sampling/sampling_logp_difference/max": 1.8052539825439453, "sampling/sampling_logp_difference/mean": 0.0048310887068510056, "step": 2737 }, { "clip_ratio/high_max": 0.006403850042261183, "clip_ratio/high_mean": 0.006403850042261183, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006403850042261183, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.25, "completions/mean_terminated_length": 97.25, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.06638350430876017, "epoch": 0.10997308912720408, "frac_reward_zero_std": 0.0, "grad_norm": 0.7919180393218994, "learning_rate": 1.7060606060606062e-06, "loss": -0.0017, "num_tokens": 6202503.0, "reward": 0.9980356693267822, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980356693267822, "reward_meter_std": 6.821734132245183e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.822366412961856e-05, "reward_total_composite_mean": 0.9980356693267822, "reward_total_composite_std": 6.821734132245183e-05, "reward_total_mean": 0.9980356693267822, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980356693267822, "rewards/meter/std": 6.821734132245183e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980356693267822, "rewards/total_composite/std": 6.821734132245183e-05, "sampling/importance_sampling_ratio/max": 1.4862256050109863, "sampling/importance_sampling_ratio/mean": 1.003057837486267, "sampling/importance_sampling_ratio/min": 0.6618210673332214, "sampling/sampling_logp_difference/max": 0.41276001930236816, "sampling/sampling_logp_difference/mean": 0.007202071137726307, "step": 2738 }, { "clip_ratio/high_max": 0.016798419412225485, "clip_ratio/high_mean": 0.016798419412225485, "clip_ratio/low_mean": 0.04031385388225317, "clip_ratio/low_min": 0.04031385388225317, "clip_ratio/region_mean": 0.057112273294478655, "completions/clipped_ratio": 0.0, "completions/max_length": 46.0, "completions/max_terminated_length": 46.0, "completions/mean_length": 44.0, "completions/mean_terminated_length": 44.0, "completions/min_length": 42.0, "completions/min_terminated_length": 42.0, "entropy": 0.21985023096203804, "epoch": 0.11001325460898903, "frac_reward_zero_std": 0.0, "grad_norm": 4.850316524505615, "learning_rate": 1.703030303030303e-06, "loss": -0.0016, "num_tokens": 6204047.0, "reward": 0.9376137256622314, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9376137256622314, "reward_meter_std": 0.004042606335133314, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004042605869472027, "reward_total_composite_mean": 0.9376137256622314, "reward_total_composite_std": 0.004042606335133314, "reward_total_mean": 0.9376137256622314, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9376137256622314, "rewards/meter/std": 0.004042606335133314, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9376137256622314, "rewards/total_composite/std": 0.004042606335133314, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9946960210800171, "sampling/importance_sampling_ratio/min": 0.13374659419059753, "sampling/sampling_logp_difference/max": 2.011808395385742, "sampling/sampling_logp_difference/mean": 0.0746898204088211, "step": 2739 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.018359567504376173, "epoch": 0.11005342009077398, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.7000000000000002e-06, "loss": 0.0, "num_tokens": 6205946.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.19320809841156, "sampling/importance_sampling_ratio/mean": 0.9993410110473633, "sampling/importance_sampling_ratio/min": 0.5584760904312134, "sampling/sampling_logp_difference/max": 0.5825433731079102, "sampling/sampling_logp_difference/mean": 0.0036742938682436943, "step": 2740 }, { "clip_ratio/high_max": 0.0012886597542092204, "clip_ratio/high_mean": 0.0012886597542092204, "clip_ratio/low_mean": 0.005348057369701564, "clip_ratio/low_min": 0.005348057369701564, "clip_ratio/region_mean": 0.006636717123910785, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 96.125, "completions/mean_terminated_length": 96.125, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.07564361020922661, "epoch": 0.11009358557255894, "frac_reward_zero_std": 0.0, "grad_norm": 1.060721755027771, "learning_rate": 1.6969696969696973e-06, "loss": -0.0202, "num_tokens": 6207995.0, "reward": 0.9955133199691772, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955133199691772, "reward_meter_std": 0.004802301991730928, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004802299663424492, "reward_total_composite_mean": 0.9955133199691772, "reward_total_composite_std": 0.004802301991730928, "reward_total_mean": 0.9955133199691772, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955133199691772, "rewards/meter/std": 0.004802301991730928, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955133199691772, "rewards/total_composite/std": 0.004802301991730928, "sampling/importance_sampling_ratio/max": 1.8030701875686646, "sampling/importance_sampling_ratio/mean": 1.0021657943725586, "sampling/importance_sampling_ratio/min": 0.5616306066513062, "sampling/sampling_logp_difference/max": 0.5894908905029297, "sampling/sampling_logp_difference/mean": 0.006459005642682314, "step": 2741 }, { "clip_ratio/high_max": 0.02870650147087872, "clip_ratio/high_mean": 0.02870650147087872, "clip_ratio/low_mean": 0.011077517876401544, "clip_ratio/low_min": 0.011077517876401544, "clip_ratio/region_mean": 0.039784019347280264, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 100.75, "completions/mean_terminated_length": 100.75, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.329579945653677, "epoch": 0.11013375105434389, "frac_reward_zero_std": 0.0, "grad_norm": 3.1250481605529785, "learning_rate": 1.6939393939393941e-06, "loss": 0.0053, "num_tokens": 6210017.0, "reward": 0.9990947842597961, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990947842597961, "reward_meter_std": 0.0001982693502213806, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00019826588686555624, "reward_total_composite_mean": 0.9990947842597961, "reward_total_composite_std": 0.0001982693502213806, "reward_total_mean": 0.9990947842597961, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990947842597961, "rewards/meter/std": 0.0001982693502213806, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990947842597961, "rewards/total_composite/std": 0.0001982693502213806, "sampling/importance_sampling_ratio/max": 1.9947516918182373, "sampling/importance_sampling_ratio/mean": 1.0069540739059448, "sampling/importance_sampling_ratio/min": 0.07659865915775299, "sampling/sampling_logp_difference/max": 2.5691757202148438, "sampling/sampling_logp_difference/mean": 0.043827299028635025, "step": 2742 }, { "clip_ratio/high_max": 0.01622902590315789, "clip_ratio/high_mean": 0.01622902590315789, "clip_ratio/low_mean": 0.014685500762425363, "clip_ratio/low_min": 0.014685500762425363, "clip_ratio/region_mean": 0.030914526665583253, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.25, "completions/mean_terminated_length": 69.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.33481148816645145, "epoch": 0.11017391653612885, "frac_reward_zero_std": 0.0, "grad_norm": 2.3616504669189453, "learning_rate": 1.6909090909090912e-06, "loss": 0.0, "num_tokens": 6211811.0, "reward": 0.9709841012954712, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9709841012954712, "reward_meter_std": 0.02855425886809826, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.028554245829582214, "reward_total_composite_mean": 0.9709841012954712, "reward_total_composite_std": 0.02855425886809826, "reward_total_mean": 0.9709841012954712, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9709841012954712, "rewards/meter/std": 0.02855425886809826, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9709841012954712, "rewards/total_composite/std": 0.02855425886809826, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0066807270050049, "sampling/importance_sampling_ratio/min": 0.3372916579246521, "sampling/sampling_logp_difference/max": 1.5528497695922852, "sampling/sampling_logp_difference/mean": 0.04137599095702171, "step": 2743 }, { "clip_ratio/high_max": 0.026779879350215197, "clip_ratio/high_mean": 0.026779879350215197, "clip_ratio/low_mean": 0.013101526303216815, "clip_ratio/low_min": 0.013101526303216815, "clip_ratio/region_mean": 0.03988140565343201, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 294.0, "completions/mean_terminated_length": 294.0, "completions/min_length": 272.0, "completions/min_terminated_length": 272.0, "entropy": 0.35994909703731537, "epoch": 0.1102140820179138, "frac_reward_zero_std": 0.0, "grad_norm": 3.2804408073425293, "learning_rate": 1.687878787878788e-06, "loss": -0.0272, "num_tokens": 6215771.0, "reward": 0.9384926557540894, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.04629101976752281, "reward_meter_mean": 0.996207058429718, "reward_meter_std": 0.0008007285068742931, "reward_repeat_penalty_mean": 0.9663312435150146, "reward_repeat_penalty_std": 0.039662934839725494, "reward_std": 0.057300180196762085, "reward_total_composite_mean": 0.9384926557540894, "reward_total_composite_std": 0.05730018764734268, "reward_total_mean": 0.9384926557540894, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.04629101976752281, "rewards/meter/mean": 0.996207058429718, "rewards/meter/std": 0.0008007285068742931, "rewards/repeat_penalty/mean": 0.9663312435150146, "rewards/repeat_penalty/std": 0.039662934839725494, "rewards/total_composite/mean": 0.9384926557540894, "rewards/total_composite/std": 0.05730018764734268, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.002259373664856, "sampling/importance_sampling_ratio/min": 0.1505059003829956, "sampling/sampling_logp_difference/max": 1.8937530517578125, "sampling/sampling_logp_difference/mean": 0.04739288613200188, "step": 2744 }, { "clip_ratio/high_max": 0.012085011578164995, "clip_ratio/high_mean": 0.012085011578164995, "clip_ratio/low_mean": 0.0029986235313117504, "clip_ratio/low_min": 0.0029986235313117504, "clip_ratio/region_mean": 0.015083635109476745, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 497.0, "completions/mean_length": 481.5, "completions/mean_terminated_length": 477.14288330078125, "completions/min_length": 458.0, "completions/min_terminated_length": 458.0, "entropy": 0.17778486385941505, "epoch": 0.11025424749969875, "frac_reward_zero_std": 0.0, "grad_norm": 0.6794636249542236, "learning_rate": 1.684848484848485e-06, "loss": -0.2843, "num_tokens": 6220927.0, "reward": 0.6496540307998657, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8515625, "reward_count_adherence_std": 0.04650149121880531, "reward_meter_mean": 0.9529277086257935, "reward_meter_std": 0.12985190749168396, "reward_repeat_penalty_mean": 0.8053357601165771, "reward_repeat_penalty_std": 0.025981293991208076, "reward_std": 0.07202339917421341, "reward_total_composite_mean": 0.6496540307998657, "reward_total_composite_std": 0.07202339172363281, "reward_total_mean": 0.6496540307998657, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8515625, "rewards/count_adherence/std": 0.04650149121880531, "rewards/meter/mean": 0.9529277086257935, "rewards/meter/std": 0.12985190749168396, "rewards/repeat_penalty/mean": 0.8053357601165771, "rewards/repeat_penalty/std": 0.025981293991208076, "rewards/total_composite/mean": 0.6496540307998657, "rewards/total_composite/std": 0.07202339172363281, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0057053565979004, "sampling/importance_sampling_ratio/min": 0.06733670830726624, "sampling/sampling_logp_difference/max": 2.698049783706665, "sampling/sampling_logp_difference/mean": 0.024653667584061623, "step": 2745 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.002097451186273247, "epoch": 0.11029441298148371, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.6818181818181819e-06, "loss": 0.0, "num_tokens": 6222631.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0122551918029785, "sampling/importance_sampling_ratio/mean": 0.9999988079071045, "sampling/importance_sampling_ratio/min": 0.9392772912979126, "sampling/sampling_logp_difference/max": 0.06264449656009674, "sampling/sampling_logp_difference/mean": 0.0003325922298245132, "step": 2746 }, { "clip_ratio/high_max": 0.020309405983425677, "clip_ratio/high_mean": 0.020309405983425677, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/region_mean": 0.02209512027911842, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.5, "completions/mean_terminated_length": 68.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2892153840512037, "epoch": 0.11033457846326866, "frac_reward_zero_std": 0.0, "grad_norm": 3.3234918117523193, "learning_rate": 1.678787878787879e-06, "loss": 0.0146, "num_tokens": 6224387.0, "reward": 0.973109781742096, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.973109781742096, "reward_meter_std": 0.031855545938014984, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.031855542212724686, "reward_total_composite_mean": 0.973109781742096, "reward_total_composite_std": 0.031855545938014984, "reward_total_mean": 0.973109781742096, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.973109781742096, "rewards/meter/std": 0.031855545938014984, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.973109781742096, "rewards/total_composite/std": 0.031855545938014984, "sampling/importance_sampling_ratio/max": 1.755409598350525, "sampling/importance_sampling_ratio/mean": 1.0069949626922607, "sampling/importance_sampling_ratio/min": 0.21433962881565094, "sampling/sampling_logp_difference/max": 1.5401935577392578, "sampling/sampling_logp_difference/mean": 0.038049403578042984, "step": 2747 }, { "clip_ratio/high_max": 0.01721590873785317, "clip_ratio/high_mean": 0.01721590873785317, "clip_ratio/low_mean": 0.005000000121071935, "clip_ratio/low_min": 0.005000000121071935, "clip_ratio/region_mean": 0.022215908858925104, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 296.75, "completions/mean_terminated_length": 296.75, "completions/min_length": 268.0, "completions/min_terminated_length": 268.0, "entropy": 0.16522958502173424, "epoch": 0.11037474394505362, "frac_reward_zero_std": 0.0, "grad_norm": 1.725966215133667, "learning_rate": 1.675757575757576e-06, "loss": 0.0236, "num_tokens": 6228377.0, "reward": 0.8153371810913086, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.890625, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9991159439086914, "reward_meter_std": 0.0003877849958371371, "reward_repeat_penalty_mean": 0.9171568751335144, "reward_repeat_penalty_std": 0.033501796424388885, "reward_std": 0.031080493703484535, "reward_total_composite_mean": 0.8153371810913086, "reward_total_composite_std": 0.031080491840839386, "reward_total_mean": 0.8153371810913086, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.890625, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9991159439086914, "rewards/meter/std": 0.0003877849958371371, "rewards/repeat_penalty/mean": 0.9171568751335144, "rewards/repeat_penalty/std": 0.033501796424388885, "rewards/total_composite/mean": 0.8153371810913086, "rewards/total_composite/std": 0.031080491840839386, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0036824941635132, "sampling/importance_sampling_ratio/min": 0.03533553332090378, "sampling/sampling_logp_difference/max": 3.3428661823272705, "sampling/sampling_logp_difference/mean": 0.025589879602193832, "step": 2748 }, { "clip_ratio/high_max": 0.019543501897715032, "clip_ratio/high_mean": 0.019543501897715032, "clip_ratio/low_mean": 0.011289363959804177, "clip_ratio/low_min": 0.011289363959804177, "clip_ratio/region_mean": 0.03083286585751921, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 101.875, "completions/mean_terminated_length": 101.875, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.36418476328253746, "epoch": 0.11041490942683857, "frac_reward_zero_std": 0.0, "grad_norm": 3.2416765689849854, "learning_rate": 1.6727272727272728e-06, "loss": -0.0163, "num_tokens": 6230560.0, "reward": 0.9238500595092773, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9238500595092773, "reward_meter_std": 0.10599032789468765, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.10599032789468765, "reward_total_composite_mean": 0.9238500595092773, "reward_total_composite_std": 0.10599032789468765, "reward_total_mean": 0.9238500595092773, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9238500595092773, "rewards/meter/std": 0.10599032789468765, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9238500595092773, "rewards/total_composite/std": 0.10599032789468765, "sampling/importance_sampling_ratio/max": 1.7313110828399658, "sampling/importance_sampling_ratio/mean": 1.008776307106018, "sampling/importance_sampling_ratio/min": 0.22782805562019348, "sampling/sampling_logp_difference/max": 1.4791641235351562, "sampling/sampling_logp_difference/mean": 0.03779112920165062, "step": 2749 }, { "clip_ratio/high_max": 0.008981169667094946, "clip_ratio/high_mean": 0.008981169667094946, "clip_ratio/low_mean": 0.0013020833721384406, "clip_ratio/low_min": 0.0013020833721384406, "clip_ratio/region_mean": 0.010283253039233387, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.0, "completions/mean_terminated_length": 97.0, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.07637950498610735, "epoch": 0.11045507490862352, "frac_reward_zero_std": 0.0, "grad_norm": 3.4270999431610107, "learning_rate": 1.6696969696969698e-06, "loss": -0.0042, "num_tokens": 6232648.0, "reward": 0.9976779818534851, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976779818534851, "reward_meter_std": 0.0009901836747303605, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009901747107505798, "reward_total_composite_mean": 0.9976779818534851, "reward_total_composite_std": 0.0009901836747303605, "reward_total_mean": 0.9976779818534851, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976779818534851, "rewards/meter/std": 0.0009901836747303605, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976779818534851, "rewards/total_composite/std": 0.0009901836747303605, "sampling/importance_sampling_ratio/max": 1.6211403608322144, "sampling/importance_sampling_ratio/mean": 1.0021347999572754, "sampling/importance_sampling_ratio/min": 0.49127906560897827, "sampling/sampling_logp_difference/max": 0.7107429504394531, "sampling/sampling_logp_difference/mean": 0.005543787498027086, "step": 2750 }, { "epoch": 0.11045507490862352, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.0, "eval_completions/max_length": 410.2307692307692, "eval_completions/max_terminated_length": 410.2307692307692, "eval_completions/mean_length": 208.81730769230768, "eval_completions/mean_terminated_length": 208.81730769230768, "eval_completions/min_length": 59.30769230769231, "eval_completions/min_terminated_length": 59.30769230769231, "eval_entropy": 0.32508696615695953, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 6232648.0, "eval_reward": 0.7168137293595535, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9510142528093778, "eval_reward_count_adherence_std": 0.06824518969425789, "eval_reward_meter_mean": 0.8095979323753943, "eval_reward_meter_std": 0.2978983329465756, "eval_reward_repeat_penalty_mean": 0.9246861888812139, "eval_reward_repeat_penalty_std": 0.09499467995304328, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7168137293595535, "eval_reward_total_composite_std": 0.2932796369378383, "eval_reward_total_mean": 0.7168137293595535, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9510142528093778, "eval_rewards/count_adherence/std": 0.06824518969425789, "eval_rewards/meter/mean": 0.8095979323753943, "eval_rewards/meter/std": 0.2978983329465756, "eval_rewards/repeat_penalty/mean": 0.9246861888812139, "eval_rewards/repeat_penalty/std": 0.09499467995304328, "eval_rewards/total_composite/mean": 0.7168137293595535, "eval_rewards/total_composite/std": 0.2932796369378383, "eval_runtime": 76.3731, "eval_samples_per_second": 1.362, "eval_sampling/importance_sampling_ratio/max": 1.4980517717508168, "eval_sampling/importance_sampling_ratio/mean": 1.0077975529890795, "eval_sampling/importance_sampling_ratio/min": 0.26402970059559894, "eval_sampling/sampling_logp_difference/max": 1.3938810641948993, "eval_sampling/sampling_logp_difference/mean": 0.03025670349597931, "eval_steps_per_second": 0.17, "step": 2750 }, { "clip_ratio/high_max": 0.029523379867896438, "clip_ratio/high_mean": 0.029523379867896438, "clip_ratio/low_mean": 0.010291178710758686, "clip_ratio/low_min": 0.010291178710758686, "clip_ratio/region_mean": 0.039814558578655124, "completions/clipped_ratio": 0.0, "completions/max_length": 276.0, "completions/max_terminated_length": 276.0, "completions/mean_length": 270.75, "completions/mean_terminated_length": 270.75, "completions/min_length": 266.0, "completions/min_terminated_length": 266.0, "entropy": 0.44462598487734795, "epoch": 0.11049524039040848, "frac_reward_zero_std": 0.0, "grad_norm": 1.9583728313446045, "learning_rate": 1.6666666666666667e-06, "loss": -0.004, "num_tokens": 6236726.0, "reward": 0.9987819194793701, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987819194793701, "reward_meter_std": 0.0005108492914587259, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005108449840918183, "reward_total_composite_mean": 0.9987819194793701, "reward_total_composite_std": 0.0005108492914587259, "reward_total_mean": 0.9987819194793701, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987819194793701, "rewards/meter/std": 0.0005108492914587259, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987819194793701, "rewards/total_composite/std": 0.0005108492914587259, "sampling/importance_sampling_ratio/max": 1.7512770891189575, "sampling/importance_sampling_ratio/mean": 1.0091570615768433, "sampling/importance_sampling_ratio/min": 0.25579512119293213, "sampling/sampling_logp_difference/max": 1.3633785247802734, "sampling/sampling_logp_difference/mean": 0.05048610270023346, "step": 2751 }, { "clip_ratio/high_max": 0.011767410207539797, "clip_ratio/high_mean": 0.011767410207539797, "clip_ratio/low_mean": 0.01261913578491658, "clip_ratio/low_min": 0.01261913578491658, "clip_ratio/region_mean": 0.024386545992456377, "completions/clipped_ratio": 0.0, "completions/max_length": 252.0, "completions/max_terminated_length": 252.0, "completions/mean_length": 237.375, "completions/mean_terminated_length": 237.375, "completions/min_length": 214.0, "completions/min_terminated_length": 214.0, "entropy": 0.41799643263220787, "epoch": 0.11053540587219343, "frac_reward_zero_std": 0.0, "grad_norm": 2.32265567779541, "learning_rate": 1.6636363636363637e-06, "loss": 0.0122, "num_tokens": 6240441.0, "reward": 0.7755623459815979, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9126553535461426, "reward_meter_std": 0.10828904807567596, "reward_repeat_penalty_mean": 0.8564560413360596, "reward_repeat_penalty_std": 0.10748697072267532, "reward_std": 0.09196322411298752, "reward_total_composite_mean": 0.7755623459815979, "reward_total_composite_std": 0.09196323156356812, "reward_total_mean": 0.7755623459815979, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9126553535461426, "rewards/meter/std": 0.10828904807567596, "rewards/repeat_penalty/mean": 0.8564560413360596, "rewards/repeat_penalty/std": 0.10748697072267532, "rewards/total_composite/mean": 0.7755623459815979, "rewards/total_composite/std": 0.09196323156356812, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007071614265442, "sampling/importance_sampling_ratio/min": 0.10543066263198853, "sampling/sampling_logp_difference/max": 2.249701738357544, "sampling/sampling_logp_difference/mean": 0.04030514135956764, "step": 2752 }, { "clip_ratio/high_max": 0.05645520123653114, "clip_ratio/high_mean": 0.05645520123653114, "clip_ratio/low_mean": 0.0029761905316263437, "clip_ratio/low_min": 0.0029761905316263437, "clip_ratio/region_mean": 0.05943139176815748, "completions/clipped_ratio": 0.0, "completions/max_length": 47.0, "completions/max_terminated_length": 47.0, "completions/mean_length": 44.5, "completions/mean_terminated_length": 44.5, "completions/min_length": 41.0, "completions/min_terminated_length": 41.0, "entropy": 0.2707332409918308, "epoch": 0.11057557135397839, "frac_reward_zero_std": 0.0, "grad_norm": 12.199896812438965, "learning_rate": 1.6606060606060605e-06, "loss": 0.0132, "num_tokens": 6241917.0, "reward": 0.930524468421936, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.930524468421936, "reward_meter_std": 0.031606100499629974, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03160610795021057, "reward_total_composite_mean": 0.930524468421936, "reward_total_composite_std": 0.031606100499629974, "reward_total_mean": 0.930524468421936, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.930524468421936, "rewards/meter/std": 0.031606100499629974, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.930524468421936, "rewards/total_composite/std": 0.031606100499629974, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9975581765174866, "sampling/importance_sampling_ratio/min": 0.10229774564504623, "sampling/sampling_logp_difference/max": 2.279867649078369, "sampling/sampling_logp_difference/mean": 0.07848232239484787, "step": 2753 }, { "clip_ratio/high_max": 0.016477052122354507, "clip_ratio/high_mean": 0.016477052122354507, "clip_ratio/low_mean": 0.017272761091589928, "clip_ratio/low_min": 0.017272761091589928, "clip_ratio/region_mean": 0.033749813213944435, "completions/clipped_ratio": 0.0, "completions/max_length": 450.0, "completions/max_terminated_length": 450.0, "completions/mean_length": 400.0, "completions/mean_terminated_length": 400.0, "completions/min_length": 363.0, "completions/min_terminated_length": 363.0, "entropy": 0.45330196619033813, "epoch": 0.11061573683576334, "frac_reward_zero_std": 0.0, "grad_norm": 1.868078351020813, "learning_rate": 1.6575757575757578e-06, "loss": 0.0001, "num_tokens": 6246685.0, "reward": 0.667221188545227, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.033064987510442734, "reward_meter_mean": 0.9454325437545776, "reward_meter_std": 0.0641726702451706, "reward_repeat_penalty_mean": 0.8041722774505615, "reward_repeat_penalty_std": 0.09782867133617401, "reward_std": 0.10969908535480499, "reward_total_composite_mean": 0.667221188545227, "reward_total_composite_std": 0.10969909280538559, "reward_total_mean": 0.667221188545227, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.033064987510442734, "rewards/meter/mean": 0.9454325437545776, "rewards/meter/std": 0.0641726702451706, "rewards/repeat_penalty/mean": 0.8041722774505615, "rewards/repeat_penalty/std": 0.09782867133617401, "rewards/total_composite/mean": 0.667221188545227, "rewards/total_composite/std": 0.10969909280538559, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0145169496536255, "sampling/importance_sampling_ratio/min": 2.045658038696274e-05, "sampling/sampling_logp_difference/max": 10.797205924987793, "sampling/sampling_logp_difference/mean": 0.048431482166051865, "step": 2754 }, { "clip_ratio/high_max": 0.0016666667070239782, "clip_ratio/high_mean": 0.0016666667070239782, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0016666667070239782, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 71.5, "completions/mean_terminated_length": 71.5, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.031015932094305754, "epoch": 0.1106559023175483, "frac_reward_zero_std": 0.0, "grad_norm": 0.17323201894760132, "learning_rate": 1.6545454545454548e-06, "loss": -0.0026, "num_tokens": 6248505.0, "reward": 0.9994492530822754, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994492530822754, "reward_meter_std": 2.2336023903335445e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.232198130514007e-05, "reward_total_composite_mean": 0.9994492530822754, "reward_total_composite_std": 2.2336023903335445e-05, "reward_total_mean": 0.9994492530822754, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994492530822754, "rewards/meter/std": 2.2336023903335445e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994492530822754, "rewards/total_composite/std": 2.2336023903335445e-05, "sampling/importance_sampling_ratio/max": 1.0643366575241089, "sampling/importance_sampling_ratio/mean": 1.0008584260940552, "sampling/importance_sampling_ratio/min": 0.5781075954437256, "sampling/sampling_logp_difference/max": 0.5479953289031982, "sampling/sampling_logp_difference/mean": 0.0045864214189350605, "step": 2755 }, { "clip_ratio/high_max": 0.02566080493852496, "clip_ratio/high_mean": 0.02566080493852496, "clip_ratio/low_mean": 0.009910162014421076, "clip_ratio/low_min": 0.009910162014421076, "clip_ratio/region_mean": 0.03557096695294604, "completions/clipped_ratio": 0.0, "completions/max_length": 213.0, "completions/max_terminated_length": 213.0, "completions/mean_length": 204.5, "completions/mean_terminated_length": 204.5, "completions/min_length": 181.0, "completions/min_terminated_length": 181.0, "entropy": 0.3736105412244797, "epoch": 0.11069606779933325, "frac_reward_zero_std": 0.0, "grad_norm": 2.2315144538879395, "learning_rate": 1.6515151515151517e-06, "loss": -0.0331, "num_tokens": 6251781.0, "reward": 0.8484304547309875, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9791666269302368, "reward_count_adherence_std": 0.0589255727827549, "reward_meter_mean": 0.9686325192451477, "reward_meter_std": 0.029917554929852486, "reward_repeat_penalty_mean": 0.8901515007019043, "reward_repeat_penalty_std": 0.10736672580242157, "reward_std": 0.13986073434352875, "reward_total_composite_mean": 0.8484304547309875, "reward_total_composite_std": 0.13986076414585114, "reward_total_mean": 0.8484304547309875, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9791666269302368, "rewards/count_adherence/std": 0.0589255727827549, "rewards/meter/mean": 0.9686325192451477, "rewards/meter/std": 0.029917554929852486, "rewards/repeat_penalty/mean": 0.8901515007019043, "rewards/repeat_penalty/std": 0.10736672580242157, "rewards/total_composite/mean": 0.8484304547309875, "rewards/total_composite/std": 0.13986076414585114, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0120221376419067, "sampling/importance_sampling_ratio/min": 0.05258508399128914, "sampling/sampling_logp_difference/max": 2.9453227519989014, "sampling/sampling_logp_difference/mean": 0.04765074700117111, "step": 2756 }, { "clip_ratio/high_max": 0.010339219472371042, "clip_ratio/high_mean": 0.010339219472371042, "clip_ratio/low_mean": 0.006716518080793321, "clip_ratio/low_min": 0.006716518080793321, "clip_ratio/region_mean": 0.017055737553164363, "completions/clipped_ratio": 0.0, "completions/max_length": 336.0, "completions/max_terminated_length": 336.0, "completions/mean_length": 332.625, "completions/mean_terminated_length": 332.625, "completions/min_length": 322.0, "completions/min_terminated_length": 322.0, "entropy": 0.19466014206409454, "epoch": 0.1107362332811182, "frac_reward_zero_std": 0.0, "grad_norm": 1.6318613290786743, "learning_rate": 1.6484848484848487e-06, "loss": 0.0142, "num_tokens": 6256090.0, "reward": 0.8277831077575684, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9090909361839294, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958105087280273, "reward_meter_std": 0.006650847382843494, "reward_repeat_penalty_mean": 0.9144736528396606, "reward_repeat_penalty_std": 0.027239451184868813, "reward_std": 0.022337697446346283, "reward_total_composite_mean": 0.8277831077575684, "reward_total_composite_std": 0.02233772911131382, "reward_total_mean": 0.8277831077575684, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9090909361839294, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958105087280273, "rewards/meter/std": 0.006650847382843494, "rewards/repeat_penalty/mean": 0.9144736528396606, "rewards/repeat_penalty/std": 0.027239451184868813, "rewards/total_composite/mean": 0.8277831077575684, "rewards/total_composite/std": 0.02233772911131382, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004244089126587, "sampling/importance_sampling_ratio/min": 0.07772251218557358, "sampling/sampling_logp_difference/max": 2.554610252380371, "sampling/sampling_logp_difference/mean": 0.02632460929453373, "step": 2757 }, { "clip_ratio/high_max": 0.010245901066809893, "clip_ratio/high_mean": 0.010245901066809893, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.010245901066809893, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.04664692562073469, "epoch": 0.11077639876290316, "frac_reward_zero_std": 0.0, "grad_norm": 2.0312271118164062, "learning_rate": 1.6454545454545455e-06, "loss": 0.0014, "num_tokens": 6257914.0, "reward": 0.9973218441009521, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973218441009521, "reward_meter_std": 5.1264036301290616e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.1269918913021684e-05, "reward_total_composite_mean": 0.9973218441009521, "reward_total_composite_std": 5.1264036301290616e-05, "reward_total_mean": 0.9973218441009521, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973218441009521, "rewards/meter/std": 5.1264036301290616e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973218441009521, "rewards/total_composite/std": 5.1264036301290616e-05, "sampling/importance_sampling_ratio/max": 1.3268663883209229, "sampling/importance_sampling_ratio/mean": 0.9973583221435547, "sampling/importance_sampling_ratio/min": 0.6021311283111572, "sampling/sampling_logp_difference/max": 0.5072799921035767, "sampling/sampling_logp_difference/mean": 0.008693864569067955, "step": 2758 }, { "clip_ratio/high_max": 0.025451545137912035, "clip_ratio/high_mean": 0.025451545137912035, "clip_ratio/low_mean": 0.006415694952011108, "clip_ratio/low_min": 0.006415694952011108, "clip_ratio/region_mean": 0.03186724008992314, "completions/clipped_ratio": 0.0, "completions/max_length": 353.0, "completions/max_terminated_length": 353.0, "completions/mean_length": 345.0, "completions/mean_terminated_length": 345.0, "completions/min_length": 321.0, "completions/min_terminated_length": 321.0, "entropy": 0.4418103024363518, "epoch": 0.11081656424468811, "frac_reward_zero_std": 0.0, "grad_norm": 1.848800539970398, "learning_rate": 1.6424242424242426e-06, "loss": -0.0188, "num_tokens": 6262570.0, "reward": 0.9773909449577332, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9861111044883728, "reward_count_adherence_std": 0.03928370773792267, "reward_meter_mean": 0.9986103773117065, "reward_meter_std": 0.000737982802093029, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_std": 0.04150520637631416, "reward_total_composite_mean": 0.9773909449577332, "reward_total_composite_std": 0.04150521755218506, "reward_total_mean": 0.9773909449577332, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9861111044883728, "rewards/count_adherence/std": 0.03928370773792267, "rewards/meter/mean": 0.9986103773117065, "rewards/meter/std": 0.000737982802093029, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.9773909449577332, "rewards/total_composite/std": 0.04150521755218506, "sampling/importance_sampling_ratio/max": 1.8393433094024658, "sampling/importance_sampling_ratio/mean": 1.0085657835006714, "sampling/importance_sampling_ratio/min": 0.15567104518413544, "sampling/sampling_logp_difference/max": 1.8600101470947266, "sampling/sampling_logp_difference/mean": 0.04865448549389839, "step": 2759 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.0025047636299859732, "epoch": 0.11085672972647306, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.6393939393939396e-06, "loss": 0.0, "num_tokens": 6263986.0, "reward": 0.9980231523513794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9980231523513794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0036016702651978, "sampling/importance_sampling_ratio/mean": 1.000309944152832, "sampling/importance_sampling_ratio/min": 0.9977269768714905, "sampling/sampling_logp_difference/max": 0.0035952008329331875, "sampling/sampling_logp_difference/mean": 0.00032912864116951823, "step": 2760 }, { "clip_ratio/high_max": 0.026291712652891874, "clip_ratio/high_mean": 0.026291712652891874, "clip_ratio/low_mean": 0.007310603512451053, "clip_ratio/low_min": 0.007310603512451053, "clip_ratio/region_mean": 0.03360231616534293, "completions/clipped_ratio": 0.0, "completions/max_length": 342.0, "completions/max_terminated_length": 342.0, "completions/mean_length": 319.375, "completions/mean_terminated_length": 319.375, "completions/min_length": 306.0, "completions/min_terminated_length": 306.0, "entropy": 0.47739846259355545, "epoch": 0.11089689520825802, "frac_reward_zero_std": 0.0, "grad_norm": 2.0906882286071777, "learning_rate": 1.6363636363636365e-06, "loss": 0.0416, "num_tokens": 6268093.0, "reward": 0.946518063545227, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.9989085793495178, "reward_meter_std": 0.00021208541875239462, "reward_repeat_penalty_mean": 0.9769607782363892, "reward_repeat_penalty_std": 0.03188912943005562, "reward_std": 0.07988038659095764, "reward_total_composite_mean": 0.946518063545227, "reward_total_composite_std": 0.07988038659095764, "reward_total_mean": 0.946518063545227, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.9989085793495178, "rewards/meter/std": 0.00021208541875239462, "rewards/repeat_penalty/mean": 0.9769607782363892, "rewards/repeat_penalty/std": 0.03188912943005562, "rewards/total_composite/mean": 0.946518063545227, "rewards/total_composite/std": 0.07988038659095764, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0140368938446045, "sampling/importance_sampling_ratio/min": 0.19611649215221405, "sampling/sampling_logp_difference/max": 1.6290464401245117, "sampling/sampling_logp_difference/mean": 0.050234466791152954, "step": 2761 }, { "clip_ratio/high_max": 0.02216856903396547, "clip_ratio/high_mean": 0.02216856903396547, "clip_ratio/low_mean": 0.0017857142956927419, "clip_ratio/low_min": 0.0017857142956927419, "clip_ratio/region_mean": 0.02395428332965821, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.75, "completions/mean_terminated_length": 67.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.2622176371514797, "epoch": 0.11093706069004297, "frac_reward_zero_std": 0.0, "grad_norm": 7.329155921936035, "learning_rate": 1.6333333333333335e-06, "loss": 0.0269, "num_tokens": 6269931.0, "reward": 0.977131724357605, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.977131724357605, "reward_meter_std": 0.06190915405750275, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.06190915033221245, "reward_total_composite_mean": 0.977131724357605, "reward_total_composite_std": 0.06190915405750275, "reward_total_mean": 0.977131724357605, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.977131724357605, "rewards/meter/std": 0.06190915405750275, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.977131724357605, "rewards/total_composite/std": 0.06190915405750275, "sampling/importance_sampling_ratio/max": 1.5230276584625244, "sampling/importance_sampling_ratio/mean": 0.9983816742897034, "sampling/importance_sampling_ratio/min": 0.2558309733867645, "sampling/sampling_logp_difference/max": 1.3632383346557617, "sampling/sampling_logp_difference/mean": 0.04112435504794121, "step": 2762 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.014110149233601987, "epoch": 0.11097722617182793, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.6303030303030303e-06, "loss": 0.0, "num_tokens": 6271724.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.6751859188079834, "sampling/importance_sampling_ratio/mean": 1.0024110078811646, "sampling/importance_sampling_ratio/min": 0.9706259965896606, "sampling/sampling_logp_difference/max": 0.5159242153167725, "sampling/sampling_logp_difference/mean": 0.002258425345644355, "step": 2763 }, { "clip_ratio/high_max": 0.00474683556240052, "clip_ratio/high_mean": 0.00474683556240052, "clip_ratio/low_mean": 0.00474833813495934, "clip_ratio/low_min": 0.00474833813495934, "clip_ratio/region_mean": 0.00949517369735986, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 79.5, "completions/mean_terminated_length": 79.5, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.20842239819467068, "epoch": 0.11101739165361288, "frac_reward_zero_std": 0.0, "grad_norm": 3.616687774658203, "learning_rate": 1.6272727272727274e-06, "loss": -0.0057, "num_tokens": 6273656.0, "reward": 0.9986147880554199, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986147880554199, "reward_meter_std": 0.0008511117775924504, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008511117775924504, "reward_total_composite_mean": 0.9986147880554199, "reward_total_composite_std": 0.0008511117775924504, "reward_total_mean": 0.9986147880554199, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986147880554199, "rewards/meter/std": 0.0008511117775924504, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986147880554199, "rewards/total_composite/std": 0.0008511117775924504, "sampling/importance_sampling_ratio/max": 1.7587852478027344, "sampling/importance_sampling_ratio/mean": 1.0039986371994019, "sampling/importance_sampling_ratio/min": 0.30414915084838867, "sampling/sampling_logp_difference/max": 1.190237045288086, "sampling/sampling_logp_difference/mean": 0.029622886329889297, "step": 2764 }, { "clip_ratio/high_max": 0.03991973074153066, "clip_ratio/high_mean": 0.03991973074153066, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.04370760964229703, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 102.25, "completions/mean_terminated_length": 102.25, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.3880934603512287, "epoch": 0.11105755713539783, "frac_reward_zero_std": 0.0, "grad_norm": 6.583384037017822, "learning_rate": 1.6242424242424242e-06, "loss": -0.0034, "num_tokens": 6275778.0, "reward": 0.9677326679229736, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9677326679229736, "reward_meter_std": 0.08840977400541306, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.08840975910425186, "reward_total_composite_mean": 0.9677326679229736, "reward_total_composite_std": 0.08840977400541306, "reward_total_mean": 0.9677326679229736, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9677326679229736, "rewards/meter/std": 0.08840977400541306, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9677326679229736, "rewards/total_composite/std": 0.08840977400541306, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0036083459854126, "sampling/importance_sampling_ratio/min": 0.21470315754413605, "sampling/sampling_logp_difference/max": 1.538498878479004, "sampling/sampling_logp_difference/mean": 0.05027589201927185, "step": 2765 }, { "clip_ratio/high_max": 0.013889989699237049, "clip_ratio/high_mean": 0.013889989699237049, "clip_ratio/low_mean": 0.008585114032030106, "clip_ratio/low_min": 0.008585114032030106, "clip_ratio/region_mean": 0.022475103731267154, "completions/clipped_ratio": 0.0, "completions/max_length": 118.0, "completions/max_terminated_length": 118.0, "completions/mean_length": 116.875, "completions/mean_terminated_length": 116.875, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.2904528006911278, "epoch": 0.11109772261718279, "frac_reward_zero_std": 0.0, "grad_norm": 2.6680572032928467, "learning_rate": 1.6212121212121213e-06, "loss": -0.0051, "num_tokens": 6278065.0, "reward": 0.9987287521362305, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987287521362305, "reward_meter_std": 0.0006380022969096899, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006380106206052005, "reward_total_composite_mean": 0.9987287521362305, "reward_total_composite_std": 0.0006380022969096899, "reward_total_mean": 0.9987287521362305, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987287521362305, "rewards/meter/std": 0.0006380022969096899, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987287521362305, "rewards/total_composite/std": 0.0006380022969096899, "sampling/importance_sampling_ratio/max": 1.6575981378555298, "sampling/importance_sampling_ratio/mean": 1.0104761123657227, "sampling/importance_sampling_ratio/min": 0.45612236857414246, "sampling/sampling_logp_difference/max": 0.7849941253662109, "sampling/sampling_logp_difference/mean": 0.02768939547240734, "step": 2766 }, { "clip_ratio/high_max": 0.005081536248326302, "clip_ratio/high_mean": 0.005081536248326302, "clip_ratio/low_mean": 0.0040465755155310035, "clip_ratio/low_min": 0.0040465755155310035, "clip_ratio/region_mean": 0.009128111763857305, "completions/clipped_ratio": 0.0, "completions/max_length": 248.0, "completions/max_terminated_length": 248.0, "completions/mean_length": 246.625, "completions/mean_terminated_length": 246.625, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.09486087318509817, "epoch": 0.11113788809896774, "frac_reward_zero_std": 0.0, "grad_norm": 1.0399538278579712, "learning_rate": 1.618181818181818e-06, "loss": 0.004, "num_tokens": 6281566.0, "reward": 0.7397112250328064, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990943074226379, "reward_meter_std": 0.0001148659794125706, "reward_repeat_penalty_mean": 0.7403846383094788, "reward_repeat_penalty_std": 0.05723259598016739, "reward_std": 0.05713575705885887, "reward_total_composite_mean": 0.7397112250328064, "reward_total_composite_std": 0.05713575705885887, "reward_total_mean": 0.7397112250328064, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990943074226379, "rewards/meter/std": 0.0001148659794125706, "rewards/repeat_penalty/mean": 0.7403846383094788, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.7397112250328064, "rewards/total_composite/std": 0.05713575705885887, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0045311450958252, "sampling/importance_sampling_ratio/min": 0.24633793532848358, "sampling/sampling_logp_difference/max": 1.4010510444641113, "sampling/sampling_logp_difference/mean": 0.014130277559161186, "step": 2767 }, { "clip_ratio/high_max": 0.004426380852237344, "clip_ratio/high_mean": 0.004426380852237344, "clip_ratio/low_mean": 0.002622719621285796, "clip_ratio/low_min": 0.002622719621285796, "clip_ratio/region_mean": 0.00704910047352314, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 141.875, "completions/mean_terminated_length": 141.875, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 0.08987067546695471, "epoch": 0.1111780535807527, "frac_reward_zero_std": 0.0, "grad_norm": 2.360358476638794, "learning_rate": 1.6151515151515153e-06, "loss": 0.0045, "num_tokens": 6283917.0, "reward": 0.8920483589172363, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991145133972168, "reward_meter_std": 0.00046132790157571435, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06576978415250778, "reward_total_composite_mean": 0.8920483589172363, "reward_total_composite_std": 0.06576977670192719, "reward_total_mean": 0.8920483589172363, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991145133972168, "rewards/meter/std": 0.00046132790157571435, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8920483589172363, "rewards/total_composite/std": 0.06576977670192719, "sampling/importance_sampling_ratio/max": 1.6703215837478638, "sampling/importance_sampling_ratio/mean": 1.0041320323944092, "sampling/importance_sampling_ratio/min": 0.3763877749443054, "sampling/sampling_logp_difference/max": 0.977135419845581, "sampling/sampling_logp_difference/mean": 0.012811451219022274, "step": 2768 }, { "clip_ratio/high_max": 0.014446925837546587, "clip_ratio/high_mean": 0.014446925837546587, "clip_ratio/low_mean": 0.010852963663637638, "clip_ratio/low_min": 0.010852963663637638, "clip_ratio/region_mean": 0.025299889501184225, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 506.0, "completions/mean_length": 501.0, "completions/mean_terminated_length": 497.3333435058594, "completions/min_length": 475.0, "completions/min_terminated_length": 475.0, "entropy": 0.37871410325169563, "epoch": 0.11121821906253765, "frac_reward_zero_std": 0.0, "grad_norm": 1.8239376544952393, "learning_rate": 1.6121212121212124e-06, "loss": 0.0647, "num_tokens": 6288893.0, "reward": 0.7339591979980469, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7573529481887817, "reward_count_adherence_std": 0.020797256380319595, "reward_meter_mean": 0.9878989458084106, "reward_meter_std": 0.03154898062348366, "reward_repeat_penalty_mean": 0.97975754737854, "reward_repeat_penalty_std": 0.021684719249606133, "reward_std": 0.050603192299604416, "reward_total_composite_mean": 0.7339591979980469, "reward_total_composite_std": 0.05060318857431412, "reward_total_mean": 0.7339591979980469, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7573529481887817, "rewards/count_adherence/std": 0.020797256380319595, "rewards/meter/mean": 0.9878989458084106, "rewards/meter/std": 0.03154898062348366, "rewards/repeat_penalty/mean": 0.97975754737854, "rewards/repeat_penalty/std": 0.021684719249606133, "rewards/total_composite/mean": 0.7339591979980469, "rewards/total_composite/std": 0.05060318857431412, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0112419128417969, "sampling/importance_sampling_ratio/min": 0.05180035158991814, "sampling/sampling_logp_difference/max": 2.9603583812713623, "sampling/sampling_logp_difference/mean": 0.05508255213499069, "step": 2769 }, { "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/region_mean": 0.0034966744715347886, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.125, "completions/mean_terminated_length": 71.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.03055504336953163, "epoch": 0.1112583845443226, "frac_reward_zero_std": 0.0, "grad_norm": 0.033189766108989716, "learning_rate": 1.6090909090909092e-06, "loss": -0.0003, "num_tokens": 6290734.0, "reward": 0.9994409084320068, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994409084320068, "reward_meter_std": 2.465912302795914e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.4565813419030746e-06, "reward_total_composite_mean": 0.9994409084320068, "reward_total_composite_std": 2.465912302795914e-06, "reward_total_mean": 0.9994409084320068, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994409084320068, "rewards/meter/std": 2.465912302795914e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994409084320068, "rewards/total_composite/std": 2.465912302795914e-06, "sampling/importance_sampling_ratio/max": 1.0757405757904053, "sampling/importance_sampling_ratio/mean": 1.0016711950302124, "sampling/importance_sampling_ratio/min": 0.7745563983917236, "sampling/sampling_logp_difference/max": 0.2554647922515869, "sampling/sampling_logp_difference/mean": 0.0038025446701794863, "step": 2770 }, { "clip_ratio/high_max": 0.023043142282404006, "clip_ratio/high_mean": 0.023043142282404006, "clip_ratio/low_mean": 0.009322975529357791, "clip_ratio/low_min": 0.009322975529357791, "clip_ratio/region_mean": 0.032366117811761796, "completions/clipped_ratio": 0.0, "completions/max_length": 139.0, "completions/max_terminated_length": 139.0, "completions/mean_length": 135.125, "completions/mean_terminated_length": 135.125, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.4557015933096409, "epoch": 0.11129855002610756, "frac_reward_zero_std": 0.0, "grad_norm": 4.650167465209961, "learning_rate": 1.6060606060606063e-06, "loss": 0.0169, "num_tokens": 6293263.0, "reward": 0.9964702725410461, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9964702725410461, "reward_meter_std": 0.003808342618867755, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.003808349836617708, "reward_total_composite_mean": 0.9964702725410461, "reward_total_composite_std": 0.003808342618867755, "reward_total_mean": 0.9964702725410461, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9964702725410461, "rewards/meter/std": 0.003808342618867755, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9964702725410461, "rewards/total_composite/std": 0.003808342618867755, "sampling/importance_sampling_ratio/max": 1.650173544883728, "sampling/importance_sampling_ratio/mean": 1.0039819478988647, "sampling/importance_sampling_ratio/min": 0.2121545374393463, "sampling/sampling_logp_difference/max": 1.5504403114318848, "sampling/sampling_logp_difference/mean": 0.04582271724939346, "step": 2771 }, { "clip_ratio/high_max": 0.00938489381223917, "clip_ratio/high_mean": 0.00938489381223917, "clip_ratio/low_mean": 0.008928571594879031, "clip_ratio/low_min": 0.008928571594879031, "clip_ratio/region_mean": 0.0183134654071182, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 67.25, "completions/mean_terminated_length": 67.25, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.2543229255825281, "epoch": 0.11133871550789251, "frac_reward_zero_std": 0.0, "grad_norm": 2.930152416229248, "learning_rate": 1.6030303030303033e-06, "loss": 0.0239, "num_tokens": 6295009.0, "reward": 0.9888796210289001, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9888796210289001, "reward_meter_std": 0.006786399520933628, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006786399520933628, "reward_total_composite_mean": 0.9888796210289001, "reward_total_composite_std": 0.006786399520933628, "reward_total_mean": 0.9888796210289001, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9888796210289001, "rewards/meter/std": 0.006786399520933628, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9888796210289001, "rewards/total_composite/std": 0.006786399520933628, "sampling/importance_sampling_ratio/max": 1.6033055782318115, "sampling/importance_sampling_ratio/mean": 1.0060391426086426, "sampling/importance_sampling_ratio/min": 0.18864330649375916, "sampling/sampling_logp_difference/max": 1.6678972244262695, "sampling/sampling_logp_difference/mean": 0.028971223160624504, "step": 2772 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.012500000186264515, "clip_ratio/low_min": 0.012500000186264515, "clip_ratio/region_mean": 0.014549180399626493, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 60.875, "completions/mean_terminated_length": 60.875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.060230674454942346, "epoch": 0.11137888098967746, "frac_reward_zero_std": 0.0, "grad_norm": 5.738377094268799, "learning_rate": 1.6000000000000001e-06, "loss": 0.0032, "num_tokens": 6296752.0, "reward": 0.9969919919967651, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969919919967651, "reward_meter_std": 0.0007282263832166791, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000728218408767134, "reward_total_composite_mean": 0.9969919919967651, "reward_total_composite_std": 0.0007282263832166791, "reward_total_mean": 0.9969919919967651, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969919919967651, "rewards/meter/std": 0.0007282263832166791, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9969919919967651, "rewards/total_composite/std": 0.0007282263832166791, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0058006048202515, "sampling/importance_sampling_ratio/min": 0.6171859502792358, "sampling/sampling_logp_difference/max": 0.8574669361114502, "sampling/sampling_logp_difference/mean": 0.01134834811091423, "step": 2773 }, { "clip_ratio/high_max": 0.01708856108598411, "clip_ratio/high_mean": 0.01708856108598411, "clip_ratio/low_mean": 0.013373102992773056, "clip_ratio/low_min": 0.013373102992773056, "clip_ratio/region_mean": 0.030461664078757167, "completions/clipped_ratio": 0.0, "completions/max_length": 183.0, "completions/max_terminated_length": 183.0, "completions/mean_length": 173.5, "completions/mean_terminated_length": 173.5, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.36277873255312443, "epoch": 0.11141904647146242, "frac_reward_zero_std": 0.0, "grad_norm": 2.424381971359253, "learning_rate": 1.5969696969696972e-06, "loss": 0.0355, "num_tokens": 6299532.0, "reward": 0.9236654043197632, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9787227511405945, "reward_meter_std": 0.03248432278633118, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.08399210125207901, "reward_std": 0.08021914213895798, "reward_total_composite_mean": 0.9236654043197632, "reward_total_composite_std": 0.08021914213895798, "reward_total_mean": 0.9236654043197632, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9787227511405945, "rewards/meter/std": 0.03248432278633118, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.08399210125207901, "rewards/total_composite/mean": 0.9236654043197632, "rewards/total_composite/std": 0.08021914213895798, "sampling/importance_sampling_ratio/max": 1.7961452007293701, "sampling/importance_sampling_ratio/mean": 1.0065571069717407, "sampling/importance_sampling_ratio/min": 0.21120309829711914, "sampling/sampling_logp_difference/max": 1.5549349784851074, "sampling/sampling_logp_difference/mean": 0.04096399247646332, "step": 2774 }, { "clip_ratio/high_max": 0.03781960904598236, "clip_ratio/high_mean": 0.03781960904598236, "clip_ratio/low_mean": 0.013254429679363966, "clip_ratio/low_min": 0.013254429679363966, "clip_ratio/region_mean": 0.05107403872534633, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 86.0, "completions/mean_terminated_length": 86.0, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.2445132229477167, "epoch": 0.11145921195324737, "frac_reward_zero_std": 0.0, "grad_norm": 9.112265586853027, "learning_rate": 1.593939393939394e-06, "loss": 0.0058, "num_tokens": 6301620.0, "reward": 0.9026405811309814, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9534580707550049, "reward_meter_std": 0.004275289364159107, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.10628911107778549, "reward_std": 0.1037164255976677, "reward_total_composite_mean": 0.9026405811309814, "reward_total_composite_std": 0.1037164255976677, "reward_total_mean": 0.9026405811309814, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9534580707550049, "rewards/meter/std": 0.004275289364159107, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.10628911107778549, "rewards/total_composite/mean": 0.9026405811309814, "rewards/total_composite/std": 0.1037164255976677, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0034648180007935, "sampling/importance_sampling_ratio/min": 0.12192168831825256, "sampling/sampling_logp_difference/max": 2.1043763160705566, "sampling/sampling_logp_difference/mean": 0.06749268621206284, "step": 2775 }, { "clip_ratio/high_max": 0.02643295784946531, "clip_ratio/high_mean": 0.02643295784946531, "clip_ratio/low_mean": 0.001923076924867928, "clip_ratio/low_min": 0.001923076924867928, "clip_ratio/region_mean": 0.02835603477433324, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.625, "completions/mean_terminated_length": 65.625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.22676505334675312, "epoch": 0.11149937743503233, "frac_reward_zero_std": 0.0, "grad_norm": 2.9178199768066406, "learning_rate": 1.590909090909091e-06, "loss": 0.0002, "num_tokens": 6303505.0, "reward": 0.9525059461593628, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9525059461593628, "reward_meter_std": 0.0829981118440628, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.082998126745224, "reward_total_composite_mean": 0.9525059461593628, "reward_total_composite_std": 0.0829981118440628, "reward_total_mean": 0.9525059461593628, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9525059461593628, "rewards/meter/std": 0.0829981118440628, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9525059461593628, "rewards/total_composite/std": 0.0829981118440628, "sampling/importance_sampling_ratio/max": 1.5501948595046997, "sampling/importance_sampling_ratio/mean": 1.003873586654663, "sampling/importance_sampling_ratio/min": 0.2848776876926422, "sampling/sampling_logp_difference/max": 1.2556953430175781, "sampling/sampling_logp_difference/mean": 0.028366295620799065, "step": 2776 }, { "clip_ratio/high_max": 0.010977057158015668, "clip_ratio/high_mean": 0.010977057158015668, "clip_ratio/low_mean": 0.0031250000465661287, "clip_ratio/low_min": 0.0031250000465661287, "clip_ratio/region_mean": 0.014102057204581797, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 79.625, "completions/mean_terminated_length": 79.625, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.22624119743704796, "epoch": 0.1115395429168173, "frac_reward_zero_std": 0.0, "grad_norm": 2.4066951274871826, "learning_rate": 1.5878787878787879e-06, "loss": -0.004, "num_tokens": 6305374.0, "reward": 0.9987850189208984, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987850189208984, "reward_meter_std": 0.0002843133988790214, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00028430239763110876, "reward_total_composite_mean": 0.9987850189208984, "reward_total_composite_std": 0.0002843133988790214, "reward_total_mean": 0.9987850189208984, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987850189208984, "rewards/meter/std": 0.0002843133988790214, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987850189208984, "rewards/total_composite/std": 0.0002843133988790214, "sampling/importance_sampling_ratio/max": 1.57016921043396, "sampling/importance_sampling_ratio/mean": 1.0044121742248535, "sampling/importance_sampling_ratio/min": 0.19254659116268158, "sampling/sampling_logp_difference/max": 1.6474170684814453, "sampling/sampling_logp_difference/mean": 0.02783510647714138, "step": 2777 }, { "clip_ratio/high_max": 0.0084732057293877, "clip_ratio/high_mean": 0.0084732057293877, "clip_ratio/low_mean": 0.0009469697251915932, "clip_ratio/low_min": 0.0009469697251915932, "clip_ratio/region_mean": 0.009420175454579294, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 132.375, "completions/mean_terminated_length": 132.375, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.051168760284781456, "epoch": 0.11157970839860225, "frac_reward_zero_std": 0.0, "grad_norm": 2.4368531703948975, "learning_rate": 1.584848484848485e-06, "loss": 0.0001, "num_tokens": 6307953.0, "reward": 0.9603220820426941, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9603220820426941, "reward_meter_std": 0.11049286276102066, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.11049285531044006, "reward_total_composite_mean": 0.9603220820426941, "reward_total_composite_std": 0.11049286276102066, "reward_total_mean": 0.9603220820426941, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9603220820426941, "rewards/meter/std": 0.11049286276102066, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9603220820426941, "rewards/total_composite/std": 0.11049286276102066, "sampling/importance_sampling_ratio/max": 1.631596565246582, "sampling/importance_sampling_ratio/mean": 1.0007283687591553, "sampling/importance_sampling_ratio/min": 0.18848192691802979, "sampling/sampling_logp_difference/max": 1.6687531471252441, "sampling/sampling_logp_difference/mean": 0.007737660314887762, "step": 2778 }, { "clip_ratio/high_max": 0.011195790022611618, "clip_ratio/high_mean": 0.011195790022611618, "clip_ratio/low_mean": 0.013203723356127739, "clip_ratio/low_min": 0.013203723356127739, "clip_ratio/region_mean": 0.024399513378739357, "completions/clipped_ratio": 0.0, "completions/max_length": 124.0, "completions/max_terminated_length": 124.0, "completions/mean_length": 123.25, "completions/mean_terminated_length": 123.25, "completions/min_length": 122.0, "completions/min_terminated_length": 122.0, "entropy": 0.20793312788009644, "epoch": 0.1116198738803872, "frac_reward_zero_std": 0.0, "grad_norm": 2.0250144004821777, "learning_rate": 1.5818181818181818e-06, "loss": 0.0001, "num_tokens": 6310483.0, "reward": 0.9977097511291504, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977097511291504, "reward_meter_std": 0.0001580109674250707, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00015800821711309254, "reward_total_composite_mean": 0.9977097511291504, "reward_total_composite_std": 0.0001580109674250707, "reward_total_mean": 0.9977097511291504, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977097511291504, "rewards/meter/std": 0.0001580109674250707, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977097511291504, "rewards/total_composite/std": 0.0001580109674250707, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0012949705123901, "sampling/importance_sampling_ratio/min": 0.21245822310447693, "sampling/sampling_logp_difference/max": 1.549009919166565, "sampling/sampling_logp_difference/mean": 0.030101941898465157, "step": 2779 }, { "clip_ratio/high_max": 0.010822767158970237, "clip_ratio/high_mean": 0.010822767158970237, "clip_ratio/low_mean": 0.0015637245669495314, "clip_ratio/low_min": 0.0015637245669495314, "clip_ratio/region_mean": 0.012386491725919768, "completions/clipped_ratio": 0.0, "completions/max_length": 355.0, "completions/max_terminated_length": 355.0, "completions/mean_length": 323.375, "completions/mean_terminated_length": 323.375, "completions/min_length": 317.0, "completions/min_terminated_length": 317.0, "entropy": 0.19129730947315693, "epoch": 0.11166003936217216, "frac_reward_zero_std": 0.0, "grad_norm": 1.3197673559188843, "learning_rate": 1.5787878787878788e-06, "loss": 0.0047, "num_tokens": 6314718.0, "reward": 0.7331855297088623, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9124999642372131, "reward_count_adherence_std": 0.0353553481400013, "reward_meter_mean": 0.9988690614700317, "reward_meter_std": 0.00020919565577059984, "reward_repeat_penalty_mean": 0.8053405284881592, "reward_repeat_penalty_std": 0.046673085540533066, "reward_std": 0.033775512129068375, "reward_total_composite_mean": 0.7331855297088623, "reward_total_composite_std": 0.03377552330493927, "reward_total_mean": 0.7331855297088623, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9124999642372131, "rewards/count_adherence/std": 0.0353553481400013, "rewards/meter/mean": 0.9988690614700317, "rewards/meter/std": 0.00020919565577059984, "rewards/repeat_penalty/mean": 0.8053405284881592, "rewards/repeat_penalty/std": 0.046673085540533066, "rewards/total_composite/mean": 0.7331855297088623, "rewards/total_composite/std": 0.03377552330493927, "sampling/importance_sampling_ratio/max": 1.8094199895858765, "sampling/importance_sampling_ratio/mean": 1.0070363283157349, "sampling/importance_sampling_ratio/min": 0.010856889188289642, "sampling/sampling_logp_difference/max": 4.522955417633057, "sampling/sampling_logp_difference/mean": 0.0219265203922987, "step": 2780 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0019741212599910796, "epoch": 0.11170020484395711, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.5757575757575759e-06, "loss": 0.0, "num_tokens": 6316374.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0020439624786377, "sampling/importance_sampling_ratio/mean": 1.0002243518829346, "sampling/importance_sampling_ratio/min": 0.9993596076965332, "sampling/sampling_logp_difference/max": 0.002041937317699194, "sampling/sampling_logp_difference/mean": 0.00022916619491297752, "step": 2781 }, { "clip_ratio/high_max": 0.02071394305676222, "clip_ratio/high_mean": 0.02071394305676222, "clip_ratio/low_mean": 0.01848576357588172, "clip_ratio/low_min": 0.01848576357588172, "clip_ratio/region_mean": 0.03919970663264394, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 101.875, "completions/mean_terminated_length": 101.875, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.3706792779266834, "epoch": 0.11174037032574206, "frac_reward_zero_std": 0.0, "grad_norm": 3.458383560180664, "learning_rate": 1.572727272727273e-06, "loss": -0.0038, "num_tokens": 6318445.0, "reward": 0.998963475227356, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998963475227356, "reward_meter_std": 0.0003066870558541268, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00030668076942674816, "reward_total_composite_mean": 0.998963475227356, "reward_total_composite_std": 0.0003066870558541268, "reward_total_mean": 0.998963475227356, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998963475227356, "rewards/meter/std": 0.0003066870558541268, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998963475227356, "rewards/total_composite/std": 0.0003066870558541268, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0039867162704468, "sampling/importance_sampling_ratio/min": 0.04619181901216507, "sampling/sampling_logp_difference/max": 3.0749526023864746, "sampling/sampling_logp_difference/mean": 0.059917621314525604, "step": 2782 }, { "clip_ratio/high_max": 0.014478484634310007, "clip_ratio/high_mean": 0.014478484634310007, "clip_ratio/low_mean": 0.05653973203152418, "clip_ratio/low_min": 0.05653973203152418, "clip_ratio/region_mean": 0.07101821666583419, "completions/clipped_ratio": 0.0, "completions/max_length": 44.0, "completions/max_terminated_length": 44.0, "completions/mean_length": 42.375, "completions/mean_terminated_length": 42.375, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.3651922196149826, "epoch": 0.11178053580752702, "frac_reward_zero_std": 0.0, "grad_norm": 190.96791076660156, "learning_rate": 1.56969696969697e-06, "loss": 0.0798, "num_tokens": 6320008.0, "reward": 0.9555168151855469, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9555168151855469, "reward_meter_std": 0.011990510858595371, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011990513652563095, "reward_total_composite_mean": 0.9555168151855469, "reward_total_composite_std": 0.011990510858595371, "reward_total_mean": 0.9555168151855469, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9555168151855469, "rewards/meter/std": 0.011990510858595371, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9555168151855469, "rewards/total_composite/std": 0.011990510858595371, "sampling/importance_sampling_ratio/max": 1.9535908699035645, "sampling/importance_sampling_ratio/mean": 0.9958891868591309, "sampling/importance_sampling_ratio/min": 0.14687815308570862, "sampling/sampling_logp_difference/max": 1.91815185546875, "sampling/sampling_logp_difference/mean": 0.08888134360313416, "step": 2783 }, { "clip_ratio/high_max": 0.010314207524061203, "clip_ratio/high_mean": 0.010314207524061203, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/region_mean": 0.018647541292011738, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 60.75, "completions/mean_terminated_length": 60.75, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.07807908114045858, "epoch": 0.11182070128931197, "frac_reward_zero_std": 0.0, "grad_norm": 5.360074043273926, "learning_rate": 1.566666666666667e-06, "loss": 0.0038, "num_tokens": 6321854.0, "reward": 0.9968235492706299, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9968235492706299, "reward_meter_std": 0.0013836961006745696, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001383687136694789, "reward_total_composite_mean": 0.9968235492706299, "reward_total_composite_std": 0.0013836961006745696, "reward_total_mean": 0.9968235492706299, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9968235492706299, "rewards/meter/std": 0.0013836961006745696, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9968235492706299, "rewards/total_composite/std": 0.0013836961006745696, "sampling/importance_sampling_ratio/max": 1.77481210231781, "sampling/importance_sampling_ratio/mean": 1.002347707748413, "sampling/importance_sampling_ratio/min": 0.19736485183238983, "sampling/sampling_logp_difference/max": 1.6227011680603027, "sampling/sampling_logp_difference/mean": 0.014301088638603687, "step": 2784 }, { "clip_ratio/high_max": 0.0054351037833839655, "clip_ratio/high_mean": 0.0054351037833839655, "clip_ratio/low_mean": 0.019023986998945475, "clip_ratio/low_min": 0.019023986998945475, "clip_ratio/region_mean": 0.02445909078232944, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 92.125, "completions/mean_terminated_length": 92.125, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.14724664017558098, "epoch": 0.11186086677109693, "frac_reward_zero_std": 0.0, "grad_norm": 2.2449779510498047, "learning_rate": 1.5636363636363638e-06, "loss": 0.0043, "num_tokens": 6324095.0, "reward": 0.9976515173912048, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976515173912048, "reward_meter_std": 0.00014370458666235209, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00014370400458574295, "reward_total_composite_mean": 0.9976515173912048, "reward_total_composite_std": 0.00014370458666235209, "reward_total_mean": 0.9976515173912048, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976515173912048, "rewards/meter/std": 0.00014370458666235209, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976515173912048, "rewards/total_composite/std": 0.00014370458666235209, "sampling/importance_sampling_ratio/max": 1.9402447938919067, "sampling/importance_sampling_ratio/mean": 1.0068784952163696, "sampling/importance_sampling_ratio/min": 0.1989528387784958, "sampling/sampling_logp_difference/max": 1.614687442779541, "sampling/sampling_logp_difference/mean": 0.02354567125439644, "step": 2785 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 36.0, "completions/mean_terminated_length": 36.0, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.01709412969648838, "epoch": 0.11190103225288188, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.5606060606060609e-06, "loss": 0.0, "num_tokens": 6325535.0, "reward": 0.9996045231819153, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9996045231819153, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.029668927192688, "sampling/importance_sampling_ratio/mean": 1.0015150308609009, "sampling/importance_sampling_ratio/min": 0.9971944093704224, "sampling/sampling_logp_difference/max": 0.029237329959869385, "sampling/sampling_logp_difference/mean": 0.0015392457135021687, "step": 2786 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.00042030583426821977, "epoch": 0.11194119773466683, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.5575757575757577e-06, "loss": 0.0, "num_tokens": 6327007.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0008776187896729, "sampling/importance_sampling_ratio/mean": 1.00003981590271, "sampling/importance_sampling_ratio/min": 0.9996870756149292, "sampling/sampling_logp_difference/max": 0.0008772250730544329, "sampling/sampling_logp_difference/mean": 4.562741378322244e-05, "step": 2787 }, { "clip_ratio/high_max": 0.03628689446486533, "clip_ratio/high_mean": 0.03628689446486533, "clip_ratio/low_mean": 0.007281553465873003, "clip_ratio/low_min": 0.007281553465873003, "clip_ratio/region_mean": 0.04356844793073833, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 100.0, "completions/mean_terminated_length": 100.0, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.349405812099576, "epoch": 0.11198136321645179, "frac_reward_zero_std": 0.0, "grad_norm": 6.924910545349121, "learning_rate": 1.5545454545454547e-06, "loss": 0.0103, "num_tokens": 6329047.0, "reward": 0.9535719156265259, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9535719156265259, "reward_meter_std": 0.12491101026535034, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.12491098791360855, "reward_total_composite_mean": 0.9535719156265259, "reward_total_composite_std": 0.12491101026535034, "reward_total_mean": 0.9535719156265259, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9535719156265259, "rewards/meter/std": 0.12491101026535034, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9535719156265259, "rewards/total_composite/std": 0.12491101026535034, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9999960064888, "sampling/importance_sampling_ratio/min": 0.30497434735298157, "sampling/sampling_logp_difference/max": 1.1875276565551758, "sampling/sampling_logp_difference/mean": 0.041653111577034, "step": 2788 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0018322104588150978, "epoch": 0.11202152869823674, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.5515151515151516e-06, "loss": 0.0, "num_tokens": 6330791.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0025111436843872, "sampling/importance_sampling_ratio/mean": 1.0002046823501587, "sampling/importance_sampling_ratio/min": 0.9999552369117737, "sampling/sampling_logp_difference/max": 0.002508021891117096, "sampling/sampling_logp_difference/mean": 0.00020510550530161709, "step": 2789 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.00035309882514411584, "epoch": 0.1120616941800217, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.5484848484848486e-06, "loss": 0.0, "num_tokens": 6332279.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0006325244903564, "sampling/importance_sampling_ratio/mean": 1.000036358833313, "sampling/importance_sampling_ratio/min": 0.999407947063446, "sampling/sampling_logp_difference/max": 0.0006322840927168727, "sampling/sampling_logp_difference/mean": 4.29271676694043e-05, "step": 2790 }, { "clip_ratio/high_max": 0.005740093300119042, "clip_ratio/high_mean": 0.005740093300119042, "clip_ratio/low_mean": 0.010924369795247912, "clip_ratio/low_min": 0.010924369795247912, "clip_ratio/region_mean": 0.016664463095366955, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.35312316194176674, "epoch": 0.11210185966180665, "frac_reward_zero_std": 0.0, "grad_norm": 4.835381984710693, "learning_rate": 1.5454545454545454e-06, "loss": 0.0171, "num_tokens": 6334035.0, "reward": 0.98065185546875, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.98065185546875, "reward_meter_std": 0.016748299822211266, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.016748299822211266, "reward_total_composite_mean": 0.98065185546875, "reward_total_composite_std": 0.016748299822211266, "reward_total_mean": 0.98065185546875, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.98065185546875, "rewards/meter/std": 0.016748299822211266, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.98065185546875, "rewards/total_composite/std": 0.016748299822211266, "sampling/importance_sampling_ratio/max": 1.7263996601104736, "sampling/importance_sampling_ratio/mean": 1.0198413133621216, "sampling/importance_sampling_ratio/min": 0.265872985124588, "sampling/sampling_logp_difference/max": 1.3247365951538086, "sampling/sampling_logp_difference/mean": 0.038696978241205215, "step": 2791 }, { "clip_ratio/high_max": 0.011832875199615955, "clip_ratio/high_mean": 0.011832875199615955, "clip_ratio/low_mean": 0.008609326789155602, "clip_ratio/low_min": 0.008609326789155602, "clip_ratio/region_mean": 0.020442201988771558, "completions/clipped_ratio": 0.0, "completions/max_length": 234.0, "completions/max_terminated_length": 234.0, "completions/mean_length": 232.5, "completions/mean_terminated_length": 232.5, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.19751299265772104, "epoch": 0.1121420251435916, "frac_reward_zero_std": 0.0, "grad_norm": 2.5946123600006104, "learning_rate": 1.5424242424242425e-06, "loss": 0.007, "num_tokens": 6337431.0, "reward": 0.8965212106704712, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.97259521484375, "reward_meter_std": 0.05070994421839714, "reward_repeat_penalty_mean": 0.9230769276618958, "reward_repeat_penalty_std": 0.041117113083601, "reward_std": 0.035956308245658875, "reward_total_composite_mean": 0.8965212106704712, "reward_total_composite_std": 0.035956308245658875, "reward_total_mean": 0.8965212106704712, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.97259521484375, "rewards/meter/std": 0.05070994421839714, "rewards/repeat_penalty/mean": 0.9230769276618958, "rewards/repeat_penalty/std": 0.041117113083601, "rewards/total_composite/mean": 0.8965212106704712, "rewards/total_composite/std": 0.035956308245658875, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0048270225524902, "sampling/importance_sampling_ratio/min": 0.05952633172273636, "sampling/sampling_logp_difference/max": 2.821336507797241, "sampling/sampling_logp_difference/mean": 0.025431320071220398, "step": 2792 }, { "clip_ratio/high_max": 0.006764258956536651, "clip_ratio/high_mean": 0.006764258956536651, "clip_ratio/low_mean": 0.01233343977946788, "clip_ratio/low_min": 0.01233343977946788, "clip_ratio/region_mean": 0.01909769873600453, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 91.75, "completions/mean_terminated_length": 91.75, "completions/min_length": 90.0, "completions/min_terminated_length": 90.0, "entropy": 0.10903155151754618, "epoch": 0.11218219062537656, "frac_reward_zero_std": 0.0, "grad_norm": 2.338566541671753, "learning_rate": 1.5393939393939395e-06, "loss": -0.0046, "num_tokens": 6339557.0, "reward": 0.9973961710929871, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973961710929871, "reward_meter_std": 0.00040277960943058133, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00040277454536408186, "reward_total_composite_mean": 0.9973961710929871, "reward_total_composite_std": 0.00040277960943058133, "reward_total_mean": 0.9973961710929871, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973961710929871, "rewards/meter/std": 0.00040277960943058133, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973961710929871, "rewards/total_composite/std": 0.00040277960943058133, "sampling/importance_sampling_ratio/max": 1.4062292575836182, "sampling/importance_sampling_ratio/mean": 0.9954391717910767, "sampling/importance_sampling_ratio/min": 0.3321254551410675, "sampling/sampling_logp_difference/max": 1.1022424697875977, "sampling/sampling_logp_difference/mean": 0.018009934574365616, "step": 2793 }, { "clip_ratio/high_max": 0.0126050099497661, "clip_ratio/high_mean": 0.0126050099497661, "clip_ratio/low_mean": 0.007144315168261528, "clip_ratio/low_min": 0.007144315168261528, "clip_ratio/region_mean": 0.019749325118027627, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.25, "completions/mean_terminated_length": 69.25, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.29199461452662945, "epoch": 0.11222235610716151, "frac_reward_zero_std": 0.0, "grad_norm": 4.996799468994141, "learning_rate": 1.5363636363636364e-06, "loss": -0.0009, "num_tokens": 6341231.0, "reward": 0.9884920120239258, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9884920120239258, "reward_meter_std": 0.01208480168133974, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.012084796093404293, "reward_total_composite_mean": 0.9884920120239258, "reward_total_composite_std": 0.01208480168133974, "reward_total_mean": 0.9884920120239258, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9884920120239258, "rewards/meter/std": 0.01208480168133974, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9884920120239258, "rewards/total_composite/std": 0.01208480168133974, "sampling/importance_sampling_ratio/max": 1.6237882375717163, "sampling/importance_sampling_ratio/mean": 1.006773591041565, "sampling/importance_sampling_ratio/min": 0.24106091260910034, "sampling/sampling_logp_difference/max": 1.4227056503295898, "sampling/sampling_logp_difference/mean": 0.03242041543126106, "step": 2794 }, { "clip_ratio/high_max": 0.01491319842170924, "clip_ratio/high_mean": 0.01491319842170924, "clip_ratio/low_mean": 0.00992763601243496, "clip_ratio/low_min": 0.00992763601243496, "clip_ratio/region_mean": 0.0248408344341442, "completions/clipped_ratio": 0.0, "completions/max_length": 131.0, "completions/max_terminated_length": 131.0, "completions/mean_length": 126.0, "completions/mean_terminated_length": 126.0, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.37405042350292206, "epoch": 0.11226252158894647, "frac_reward_zero_std": 0.0, "grad_norm": 3.0096166133880615, "learning_rate": 1.5333333333333334e-06, "loss": 0.0081, "num_tokens": 6343711.0, "reward": 0.9772056937217712, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9772056937217712, "reward_meter_std": 0.017515793442726135, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.017515793442726135, "reward_total_composite_mean": 0.9772056937217712, "reward_total_composite_std": 0.017515793442726135, "reward_total_mean": 0.9772056937217712, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9772056937217712, "rewards/meter/std": 0.017515793442726135, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9772056937217712, "rewards/total_composite/std": 0.017515793442726135, "sampling/importance_sampling_ratio/max": 1.8909732103347778, "sampling/importance_sampling_ratio/mean": 1.0098986625671387, "sampling/importance_sampling_ratio/min": 0.1630331426858902, "sampling/sampling_logp_difference/max": 1.8138017654418945, "sampling/sampling_logp_difference/mean": 0.04109743982553482, "step": 2795 }, { "clip_ratio/high_max": 0.019141072407364845, "clip_ratio/high_mean": 0.019141072407364845, "clip_ratio/low_mean": 0.015304939355701208, "clip_ratio/low_min": 0.015304939355701208, "clip_ratio/region_mean": 0.03444601176306605, "completions/clipped_ratio": 0.0, "completions/max_length": 476.0, "completions/max_terminated_length": 476.0, "completions/mean_length": 451.25, "completions/mean_terminated_length": 451.25, "completions/min_length": 425.0, "completions/min_terminated_length": 425.0, "entropy": 0.47964709997177124, "epoch": 0.11230268707073142, "frac_reward_zero_std": 0.0, "grad_norm": 1.5088597536087036, "learning_rate": 1.5303030303030302e-06, "loss": -0.029, "num_tokens": 6349489.0, "reward": 0.7476283311843872, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7666666507720947, "reward_count_adherence_std": 0.035634830594062805, "reward_meter_mean": 0.9978024363517761, "reward_meter_std": 0.0035957051441073418, "reward_repeat_penalty_mean": 0.9777432680130005, "reward_repeat_penalty_std": 0.023832013830542564, "reward_std": 0.031429413706064224, "reward_total_composite_mean": 0.7476283311843872, "reward_total_composite_std": 0.03142942115664482, "reward_total_mean": 0.7476283311843872, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7666666507720947, "rewards/count_adherence/std": 0.035634830594062805, "rewards/meter/mean": 0.9978024363517761, "rewards/meter/std": 0.0035957051441073418, "rewards/repeat_penalty/mean": 0.9777432680130005, "rewards/repeat_penalty/std": 0.023832013830542564, "rewards/total_composite/mean": 0.7476283311843872, "rewards/total_composite/std": 0.03142942115664482, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0113346576690674, "sampling/importance_sampling_ratio/min": 0.005958050023764372, "sampling/sampling_logp_difference/max": 5.123012065887451, "sampling/sampling_logp_difference/mean": 0.05049360170960426, "step": 2796 }, { "clip_ratio/high_max": 0.013506086892448366, "clip_ratio/high_mean": 0.013506086892448366, "clip_ratio/low_mean": 0.009413161547854543, "clip_ratio/low_min": 0.009413161547854543, "clip_ratio/region_mean": 0.02291924844030291, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.21964257210493088, "epoch": 0.11234285255251637, "frac_reward_zero_std": 0.0, "grad_norm": 2.2734148502349854, "learning_rate": 1.5272727272727275e-06, "loss": -0.0044, "num_tokens": 6351421.0, "reward": 0.9991590976715088, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991590976715088, "reward_meter_std": 0.0001742392487358302, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00017422223754692823, "reward_total_composite_mean": 0.9991590976715088, "reward_total_composite_std": 0.0001742392487358302, "reward_total_mean": 0.9991590976715088, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991590976715088, "rewards/meter/std": 0.0001742392487358302, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991590976715088, "rewards/total_composite/std": 0.0001742392487358302, "sampling/importance_sampling_ratio/max": 1.4288572072982788, "sampling/importance_sampling_ratio/mean": 1.0024892091751099, "sampling/importance_sampling_ratio/min": 0.1897573173046112, "sampling/sampling_logp_difference/max": 1.6620092391967773, "sampling/sampling_logp_difference/mean": 0.028453955426812172, "step": 2797 }, { "clip_ratio/high_max": 0.0047420773771591485, "clip_ratio/high_mean": 0.0047420773771591485, "clip_ratio/low_mean": 0.009618075448088348, "clip_ratio/low_min": 0.009618075448088348, "clip_ratio/region_mean": 0.014360152825247496, "completions/clipped_ratio": 0.0, "completions/max_length": 134.0, "completions/max_terminated_length": 134.0, "completions/mean_length": 131.375, "completions/mean_terminated_length": 131.375, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.06868221471086144, "epoch": 0.11238301803430133, "frac_reward_zero_std": 0.0, "grad_norm": 1.5468372106552124, "learning_rate": 1.5242424242424245e-06, "loss": -0.0027, "num_tokens": 6353888.0, "reward": 0.9992819428443909, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992819428443909, "reward_meter_std": 0.00023744572536088526, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00023744485224597156, "reward_total_composite_mean": 0.9992819428443909, "reward_total_composite_std": 0.00023744572536088526, "reward_total_mean": 0.9992819428443909, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992819428443909, "rewards/meter/std": 0.00023744572536088526, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992819428443909, "rewards/total_composite/std": 0.00023744572536088526, "sampling/importance_sampling_ratio/max": 1.7840969562530518, "sampling/importance_sampling_ratio/mean": 1.000964641571045, "sampling/importance_sampling_ratio/min": 0.28168606758117676, "sampling/sampling_logp_difference/max": 1.2669620513916016, "sampling/sampling_logp_difference/mean": 0.01363647636026144, "step": 2798 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.11242318351608628, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 1.5212121212121214e-06, "loss": 0.0, "num_tokens": 6355832.0, "reward": 0.5374298095703125, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.71875, "reward_count_adherence_std": 0.025877464562654495, "reward_meter_mean": 0.9562926292419434, "reward_meter_std": 0.0762549340724945, "reward_repeat_penalty_mean": 0.9028632640838623, "reward_repeat_penalty_std": 0.0416770875453949, "reward_std": 0.2192150205373764, "reward_total_composite_mean": 0.5374298095703125, "reward_total_composite_std": 0.2192150205373764, "reward_total_mean": 0.5374298095703125, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.71875, "rewards/count_adherence/std": 0.025877464562654495, "rewards/meter/mean": 0.9562926292419434, "rewards/meter/std": 0.0762549340724945, "rewards/repeat_penalty/mean": 0.9028632640838623, "rewards/repeat_penalty/std": 0.0416770875453949, "rewards/total_composite/mean": 0.5374298095703125, "rewards/total_composite/std": 0.2192150205373764, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 2799 }, { "clip_ratio/high_max": 0.027406520443037152, "clip_ratio/high_mean": 0.027406520443037152, "clip_ratio/low_mean": 0.00767904706299305, "clip_ratio/low_min": 0.00767904706299305, "clip_ratio/region_mean": 0.0350855675060302, "completions/clipped_ratio": 0.0, "completions/max_length": 292.0, "completions/max_terminated_length": 292.0, "completions/mean_length": 271.0, "completions/mean_terminated_length": 271.0, "completions/min_length": 257.0, "completions/min_terminated_length": 257.0, "entropy": 0.4257059842348099, "epoch": 0.11246334899787123, "frac_reward_zero_std": 0.0, "grad_norm": 2.547086715698242, "learning_rate": 1.5181818181818184e-06, "loss": 0.0229, "num_tokens": 6359592.0, "reward": 0.8224738240242004, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9248241186141968, "reward_meter_std": 0.17904320359230042, "reward_repeat_penalty_mean": 0.8916666507720947, "reward_repeat_penalty_std": 0.08683133870363235, "reward_std": 0.1752013862133026, "reward_total_composite_mean": 0.8224738240242004, "reward_total_composite_std": 0.1752014011144638, "reward_total_mean": 0.8224738240242004, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9248241186141968, "rewards/meter/std": 0.17904320359230042, "rewards/repeat_penalty/mean": 0.8916666507720947, "rewards/repeat_penalty/std": 0.08683133870363235, "rewards/total_composite/mean": 0.8224738240242004, "rewards/total_composite/std": 0.1752014011144638, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0097815990447998, "sampling/importance_sampling_ratio/min": 0.2080799639225006, "sampling/sampling_logp_difference/max": 1.5698328018188477, "sampling/sampling_logp_difference/mean": 0.04360240325331688, "step": 2800 }, { "epoch": 0.11246334899787123, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/max_length": 418.84615384615387, "eval_completions/max_terminated_length": 381.15384615384613, "eval_completions/mean_length": 210.54807692307693, "eval_completions/mean_terminated_length": 200.53846271221454, "eval_completions/min_length": 60.69230769230769, "eval_completions/min_terminated_length": 60.69230769230769, "eval_entropy": 0.38292053456489855, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 6359592.0, "eval_reward": 0.717106791642996, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.9559012926541842, "eval_reward_count_adherence_std": 0.06501825956197885, "eval_reward_meter_mean": 0.7971470860334543, "eval_reward_meter_std": 0.31754180788993835, "eval_reward_repeat_penalty_mean": 0.9524615177741418, "eval_reward_repeat_penalty_std": 0.07457881277570358, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.717106791642996, "eval_reward_total_composite_std": 0.329087950862371, "eval_reward_total_mean": 0.717106791642996, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.9559012926541842, "eval_rewards/count_adherence/std": 0.06501825956197885, "eval_rewards/meter/mean": 0.7971470860334543, "eval_rewards/meter/std": 0.31754180788993835, "eval_rewards/repeat_penalty/mean": 0.9524615177741418, "eval_rewards/repeat_penalty/std": 0.07457881277570358, "eval_rewards/total_composite/mean": 0.717106791642996, "eval_rewards/total_composite/std": 0.329087950862371, "eval_runtime": 78.1168, "eval_samples_per_second": 1.331, "eval_sampling/importance_sampling_ratio/max": 1.5328054244701679, "eval_sampling/importance_sampling_ratio/mean": 1.0094741124373217, "eval_sampling/importance_sampling_ratio/min": 0.32078430171196276, "eval_sampling/sampling_logp_difference/max": 1.1617195422832782, "eval_sampling/sampling_logp_difference/mean": 0.03292630620014209, "eval_steps_per_second": 0.166, "step": 2800 }, { "clip_ratio/high_max": 0.016170406714081764, "clip_ratio/high_mean": 0.016170406714081764, "clip_ratio/low_mean": 0.004123763646930456, "clip_ratio/low_min": 0.004123763646930456, "clip_ratio/region_mean": 0.02029417036101222, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 123.0, "completions/mean_terminated_length": 123.0, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.17738944850862026, "epoch": 0.11250351447965619, "frac_reward_zero_std": 0.0, "grad_norm": 2.5363826751708984, "learning_rate": 1.5151515151515152e-06, "loss": -0.0039, "num_tokens": 6362056.0, "reward": 0.9619686603546143, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976023435592651, "reward_meter_std": 0.0004385724605526775, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06589511036872864, "reward_total_composite_mean": 0.9619686603546143, "reward_total_composite_std": 0.06589511781930923, "reward_total_mean": 0.9619686603546143, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976023435592651, "rewards/meter/std": 0.0004385724605526775, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9619686603546143, "rewards/total_composite/std": 0.06589511781930923, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0046809911727905, "sampling/importance_sampling_ratio/min": 0.3919609487056732, "sampling/sampling_logp_difference/max": 1.1188087463378906, "sampling/sampling_logp_difference/mean": 0.022641444578766823, "step": 2801 }, { "clip_ratio/high_max": 0.03208270261529833, "clip_ratio/high_mean": 0.03208270261529833, "clip_ratio/low_mean": 0.0019379844889044762, "clip_ratio/low_min": 0.0019379844889044762, "clip_ratio/region_mean": 0.03402068710420281, "completions/clipped_ratio": 0.0, "completions/max_length": 147.0, "completions/max_terminated_length": 147.0, "completions/mean_length": 135.375, "completions/mean_terminated_length": 135.375, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.4941607192158699, "epoch": 0.11254367996144114, "frac_reward_zero_std": 0.0, "grad_norm": 3.6220641136169434, "learning_rate": 1.5121212121212123e-06, "loss": -0.0145, "num_tokens": 6364387.0, "reward": 0.8960581421852112, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8960581421852112, "reward_meter_std": 0.2624220848083496, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2624220848083496, "reward_total_composite_mean": 0.8960581421852112, "reward_total_composite_std": 0.2624220848083496, "reward_total_mean": 0.8960581421852112, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8960581421852112, "rewards/meter/std": 0.2624220848083496, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8960581421852112, "rewards/total_composite/std": 0.2624220848083496, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0152724981307983, "sampling/importance_sampling_ratio/min": 0.35183805227279663, "sampling/sampling_logp_difference/max": 1.0952783823013306, "sampling/sampling_logp_difference/mean": 0.049931738525629044, "step": 2802 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0011047264706576243, "epoch": 0.1125838454432261, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.5090909090909091e-06, "loss": 0.0, "num_tokens": 6366291.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0050559043884277, "sampling/importance_sampling_ratio/mean": 1.0000933408737183, "sampling/importance_sampling_ratio/min": 0.9940345287322998, "sampling/sampling_logp_difference/max": 0.00598335824906826, "sampling/sampling_logp_difference/mean": 0.00012427744513843209, "step": 2803 }, { "clip_ratio/high_max": 0.026202646549791098, "clip_ratio/high_mean": 0.026202646549791098, "clip_ratio/low_mean": 0.010395145742222667, "clip_ratio/low_min": 0.010395145742222667, "clip_ratio/region_mean": 0.036597792292013764, "completions/clipped_ratio": 0.0, "completions/max_length": 241.0, "completions/max_terminated_length": 241.0, "completions/mean_length": 235.125, "completions/mean_terminated_length": 235.125, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.46530257537961006, "epoch": 0.11262401092501105, "frac_reward_zero_std": 0.0, "grad_norm": 2.9486567974090576, "learning_rate": 1.5060606060606062e-06, "loss": 0.0106, "num_tokens": 6369524.0, "reward": 0.9352947473526001, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9466480016708374, "reward_meter_std": 0.14826621115207672, "reward_repeat_penalty_mean": 0.9886363744735718, "reward_repeat_penalty_std": 0.03214120864868164, "reward_std": 0.14715023338794708, "reward_total_composite_mean": 0.9352947473526001, "reward_total_composite_std": 0.14715024828910828, "reward_total_mean": 0.9352947473526001, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9466480016708374, "rewards/meter/std": 0.14826621115207672, "rewards/repeat_penalty/mean": 0.9886363744735718, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.9352947473526001, "rewards/total_composite/std": 0.14715024828910828, "sampling/importance_sampling_ratio/max": 1.872164011001587, "sampling/importance_sampling_ratio/mean": 1.0136529207229614, "sampling/importance_sampling_ratio/min": 0.27005690336227417, "sampling/sampling_logp_difference/max": 1.3091225624084473, "sampling/sampling_logp_difference/mean": 0.04945862293243408, "step": 2804 }, { "clip_ratio/high_max": 0.029976313933730125, "clip_ratio/high_mean": 0.029976313933730125, "clip_ratio/low_mean": 0.0063775512389838696, "clip_ratio/low_min": 0.0063775512389838696, "clip_ratio/region_mean": 0.036353865172713995, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 100.0, "completions/mean_terminated_length": 100.0, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.3125297725200653, "epoch": 0.112664176406796, "frac_reward_zero_std": 0.0, "grad_norm": 4.279446125030518, "learning_rate": 1.5030303030303032e-06, "loss": -0.0011, "num_tokens": 6371724.0, "reward": 0.9728400707244873, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975219964981079, "reward_meter_std": 0.0041835736483335495, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07395300269126892, "reward_total_composite_mean": 0.9728400707244873, "reward_total_composite_std": 0.07395301014184952, "reward_total_mean": 0.9728400707244873, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975219964981079, "rewards/meter/std": 0.0041835736483335495, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9728400707244873, "rewards/total_composite/std": 0.07395301014184952, "sampling/importance_sampling_ratio/max": 1.6153602600097656, "sampling/importance_sampling_ratio/mean": 1.008768081665039, "sampling/importance_sampling_ratio/min": 0.2278643399477005, "sampling/sampling_logp_difference/max": 1.4790048599243164, "sampling/sampling_logp_difference/mean": 0.03659245744347572, "step": 2805 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0018152309057768434, "epoch": 0.11270434188858096, "frac_reward_zero_std": 0.0, "grad_norm": 0.8794472813606262, "learning_rate": 1.5e-06, "loss": -0.0041, "num_tokens": 6373380.0, "reward": 0.7413170337677002, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7413170337677002, "reward_meter_std": 0.13101767003536224, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.13101768493652344, "reward_total_composite_mean": 0.7413170337677002, "reward_total_composite_std": 0.13101767003536224, "reward_total_mean": 0.7413170337677002, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7413170337677002, "rewards/meter/std": 0.13101767003536224, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7413170337677002, "rewards/total_composite/std": 0.13101767003536224, "sampling/importance_sampling_ratio/max": 1.0023080110549927, "sampling/importance_sampling_ratio/mean": 0.9984825849533081, "sampling/importance_sampling_ratio/min": 0.24254991114139557, "sampling/sampling_logp_difference/max": 1.4165477752685547, "sampling/sampling_logp_difference/mean": 0.0035149240866303444, "step": 2806 }, { "clip_ratio/high_max": 0.01668829272966832, "clip_ratio/high_mean": 0.01668829272966832, "clip_ratio/low_mean": 0.004273504368029535, "clip_ratio/low_min": 0.004273504368029535, "clip_ratio/region_mean": 0.020961797097697854, "completions/clipped_ratio": 0.0, "completions/max_length": 122.0, "completions/max_terminated_length": 122.0, "completions/mean_length": 118.875, "completions/mean_terminated_length": 118.875, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.30837439745664597, "epoch": 0.11274450737036591, "frac_reward_zero_std": 0.0, "grad_norm": 1.6524964570999146, "learning_rate": 1.496969696969697e-06, "loss": -0.0079, "num_tokens": 6375595.0, "reward": 0.9989545345306396, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989545345306396, "reward_meter_std": 0.00029157428070902824, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00029156444361433387, "reward_total_composite_mean": 0.9989545345306396, "reward_total_composite_std": 0.00029157428070902824, "reward_total_mean": 0.9989545345306396, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989545345306396, "rewards/meter/std": 0.00029157428070902824, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989545345306396, "rewards/total_composite/std": 0.00029157428070902824, "sampling/importance_sampling_ratio/max": 1.6717981100082397, "sampling/importance_sampling_ratio/mean": 1.0066179037094116, "sampling/importance_sampling_ratio/min": 0.31599104404449463, "sampling/sampling_logp_difference/max": 1.1520414352416992, "sampling/sampling_logp_difference/mean": 0.03430765122175217, "step": 2807 }, { "clip_ratio/high_max": 0.00562528264708817, "clip_ratio/high_mean": 0.00562528264708817, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.007519222097471356, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.25, "completions/mean_terminated_length": 66.25, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.022154160542413592, "epoch": 0.11278467285215087, "frac_reward_zero_std": 0.0, "grad_norm": 0.13850678503513336, "learning_rate": 1.493939393939394e-06, "loss": -0.0002, "num_tokens": 6377373.0, "reward": 0.9981439709663391, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981439709663391, "reward_meter_std": 1.6234846043516882e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.6221796613535844e-05, "reward_total_composite_mean": 0.9981439709663391, "reward_total_composite_std": 1.6234846043516882e-05, "reward_total_mean": 0.9981439709663391, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981439709663391, "rewards/meter/std": 1.6234846043516882e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981439709663391, "rewards/total_composite/std": 1.6234846043516882e-05, "sampling/importance_sampling_ratio/max": 1.0641453266143799, "sampling/importance_sampling_ratio/mean": 0.9976470470428467, "sampling/importance_sampling_ratio/min": 0.46628665924072266, "sampling/sampling_logp_difference/max": 0.7629547119140625, "sampling/sampling_logp_difference/mean": 0.006040393374860287, "step": 2808 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.015775780542753637, "epoch": 0.11282483833393582, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.490909090909091e-06, "loss": 0.0, "num_tokens": 6379350.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0592564344406128, "sampling/importance_sampling_ratio/mean": 0.9995959401130676, "sampling/importance_sampling_ratio/min": 0.7136077284812927, "sampling/sampling_logp_difference/max": 0.33742189407348633, "sampling/sampling_logp_difference/mean": 0.0026704377960413694, "step": 2809 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.0013478934415616095, "epoch": 0.11286500381572077, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.4878787878787878e-06, "loss": 0.0, "num_tokens": 6380838.0, "reward": 0.9943599104881287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9943599104881287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0028506517410278, "sampling/importance_sampling_ratio/mean": 1.0001628398895264, "sampling/importance_sampling_ratio/min": 0.9998530745506287, "sampling/sampling_logp_difference/max": 0.0028465576469898224, "sampling/sampling_logp_difference/mean": 0.00016406136273872107, "step": 2810 }, { "clip_ratio/high_max": 0.013288452988490462, "clip_ratio/high_mean": 0.013288452988490462, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.015182392438873649, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.25444527715444565, "epoch": 0.11290516929750573, "frac_reward_zero_std": 0.0, "grad_norm": 4.882931232452393, "learning_rate": 1.484848484848485e-06, "loss": 0.0043, "num_tokens": 6382557.0, "reward": 0.9871849417686462, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9871849417686462, "reward_meter_std": 0.011940443888306618, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011940454132854939, "reward_total_composite_mean": 0.9871849417686462, "reward_total_composite_std": 0.011940443888306618, "reward_total_mean": 0.9871849417686462, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9871849417686462, "rewards/meter/std": 0.011940443888306618, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9871849417686462, "rewards/total_composite/std": 0.011940443888306618, "sampling/importance_sampling_ratio/max": 1.5272610187530518, "sampling/importance_sampling_ratio/mean": 1.0042191743850708, "sampling/importance_sampling_ratio/min": 0.3484816551208496, "sampling/sampling_logp_difference/max": 1.0541696548461914, "sampling/sampling_logp_difference/mean": 0.029961178079247475, "step": 2811 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.0012169194815214723, "epoch": 0.11294533477929068, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.481818181818182e-06, "loss": 0.0, "num_tokens": 6383901.0, "reward": 0.9943599104881287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9943599104881287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0031764507293701, "sampling/importance_sampling_ratio/mean": 1.0001628398895264, "sampling/importance_sampling_ratio/min": 0.9999763369560242, "sampling/sampling_logp_difference/max": 0.0031713712960481644, "sampling/sampling_logp_difference/mean": 0.00016283367585856467, "step": 2812 }, { "clip_ratio/high_max": 0.02115830732509494, "clip_ratio/high_mean": 0.02115830732509494, "clip_ratio/low_mean": 0.008746044244617224, "clip_ratio/low_min": 0.008746044244617224, "clip_ratio/region_mean": 0.029904351569712162, "completions/clipped_ratio": 0.0, "completions/max_length": 438.0, "completions/max_terminated_length": 438.0, "completions/mean_length": 402.375, "completions/mean_terminated_length": 402.375, "completions/min_length": 390.0, "completions/min_terminated_length": 390.0, "entropy": 0.43715671822428703, "epoch": 0.11298550026107564, "frac_reward_zero_std": 0.0, "grad_norm": 1.4151586294174194, "learning_rate": 1.478787878787879e-06, "loss": -0.0216, "num_tokens": 6388632.0, "reward": 0.7628299593925476, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7788461446762085, "reward_count_adherence_std": 0.027196412906050682, "reward_meter_mean": 0.9989125728607178, "reward_meter_std": 0.0006985744112171233, "reward_repeat_penalty_mean": 0.9802631139755249, "reward_repeat_penalty_std": 0.027239417657256126, "reward_std": 0.038778405636548996, "reward_total_composite_mean": 0.7628299593925476, "reward_total_composite_std": 0.0387784019112587, "reward_total_mean": 0.7628299593925476, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7788461446762085, "rewards/count_adherence/std": 0.027196412906050682, "rewards/meter/mean": 0.9989125728607178, "rewards/meter/std": 0.0006985744112171233, "rewards/repeat_penalty/mean": 0.9802631139755249, "rewards/repeat_penalty/std": 0.027239417657256126, "rewards/total_composite/mean": 0.7628299593925476, "rewards/total_composite/std": 0.0387784019112587, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010063886642456, "sampling/importance_sampling_ratio/min": 0.1342097669839859, "sampling/sampling_logp_difference/max": 2.0083513259887695, "sampling/sampling_logp_difference/mean": 0.04369629919528961, "step": 2813 }, { "clip_ratio/high_max": 0.008882812689989805, "clip_ratio/high_mean": 0.008882812689989805, "clip_ratio/low_mean": 0.0029606895986944437, "clip_ratio/low_min": 0.0029606895986944437, "clip_ratio/region_mean": 0.011843502288684249, "completions/clipped_ratio": 0.0, "completions/max_length": 128.0, "completions/max_terminated_length": 128.0, "completions/mean_length": 127.375, "completions/mean_terminated_length": 127.375, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.11378153134137392, "epoch": 0.11302566574286059, "frac_reward_zero_std": 0.0, "grad_norm": 1.2337918281555176, "learning_rate": 1.475757575757576e-06, "loss": -0.0018, "num_tokens": 6391131.0, "reward": 0.9428157210350037, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9961709976196289, "reward_meter_std": 0.0034113116562366486, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07388003170490265, "reward_total_composite_mean": 0.9428157210350037, "reward_total_composite_std": 0.07388003170490265, "reward_total_mean": 0.9428157210350037, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9961709976196289, "rewards/meter/std": 0.0034113116562366486, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9428157210350037, "rewards/total_composite/std": 0.07388003170490265, "sampling/importance_sampling_ratio/max": 1.6902966499328613, "sampling/importance_sampling_ratio/mean": 1.0020824670791626, "sampling/importance_sampling_ratio/min": 0.1707528531551361, "sampling/sampling_logp_difference/max": 1.767538070678711, "sampling/sampling_logp_difference/mean": 0.012026426382362843, "step": 2814 }, { "clip_ratio/high_max": 0.006330128293484449, "clip_ratio/high_mean": 0.006330128293484449, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006330128293484449, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 40.625, "completions/mean_terminated_length": 40.625, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.0865395087748766, "epoch": 0.11306583122464554, "frac_reward_zero_std": 0.0, "grad_norm": 1.3737421035766602, "learning_rate": 1.4727272727272728e-06, "loss": 0.013, "num_tokens": 6392648.0, "reward": 0.99871426820755, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99871426820755, "reward_meter_std": 0.0005067276651971042, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005067184683866799, "reward_total_composite_mean": 0.99871426820755, "reward_total_composite_std": 0.0005067276651971042, "reward_total_mean": 0.99871426820755, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99871426820755, "rewards/meter/std": 0.0005067276651971042, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99871426820755, "rewards/total_composite/std": 0.0005067276651971042, "sampling/importance_sampling_ratio/max": 1.4657316207885742, "sampling/importance_sampling_ratio/mean": 0.9997342824935913, "sampling/importance_sampling_ratio/min": 0.4286504089832306, "sampling/sampling_logp_difference/max": 0.8471136093139648, "sampling/sampling_logp_difference/mean": 0.01035021897405386, "step": 2815 }, { "clip_ratio/high_max": 0.022014267276972532, "clip_ratio/high_mean": 0.022014267276972532, "clip_ratio/low_mean": 0.01341109094209969, "clip_ratio/low_min": 0.01341109094209969, "clip_ratio/region_mean": 0.03542535821907222, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 102.125, "completions/mean_terminated_length": 102.125, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.33477963879704475, "epoch": 0.1131059967064305, "frac_reward_zero_std": 0.0, "grad_norm": 3.449428081512451, "learning_rate": 1.4696969696969698e-06, "loss": 0.0064, "num_tokens": 6394769.0, "reward": 0.9987610578536987, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987610578536987, "reward_meter_std": 0.0006801035488024354, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006801035488024354, "reward_total_composite_mean": 0.9987610578536987, "reward_total_composite_std": 0.0006801035488024354, "reward_total_mean": 0.9987610578536987, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987610578536987, "rewards/meter/std": 0.0006801035488024354, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987610578536987, "rewards/total_composite/std": 0.0006801035488024354, "sampling/importance_sampling_ratio/max": 1.9102437496185303, "sampling/importance_sampling_ratio/mean": 1.0010402202606201, "sampling/importance_sampling_ratio/min": 0.29644152522087097, "sampling/sampling_logp_difference/max": 1.2159053087234497, "sampling/sampling_logp_difference/mean": 0.03967723622918129, "step": 2816 }, { "clip_ratio/high_max": 0.01860184350516647, "clip_ratio/high_mean": 0.01860184350516647, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/region_mean": 0.022333186701871455, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.15694944839924574, "epoch": 0.11314616218821545, "frac_reward_zero_std": 0.0, "grad_norm": 4.832633972167969, "learning_rate": 1.4666666666666669e-06, "loss": 0.0051, "num_tokens": 6396562.0, "reward": 0.9991535544395447, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991535544395447, "reward_meter_std": 0.00035637259134091437, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00035637334804050624, "reward_total_composite_mean": 0.9991535544395447, "reward_total_composite_std": 0.00035637259134091437, "reward_total_mean": 0.9991535544395447, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991535544395447, "rewards/meter/std": 0.00035637259134091437, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991535544395447, "rewards/total_composite/std": 0.00035637259134091437, "sampling/importance_sampling_ratio/max": 1.4108344316482544, "sampling/importance_sampling_ratio/mean": 0.9987229108810425, "sampling/importance_sampling_ratio/min": 0.0025564145762473345, "sampling/sampling_logp_difference/max": 5.969149589538574, "sampling/sampling_logp_difference/mean": 0.03295252099633217, "step": 2817 }, { "clip_ratio/high_max": 0.007499999832361937, "clip_ratio/high_mean": 0.007499999832361937, "clip_ratio/low_mean": 0.0075952449114993215, "clip_ratio/low_min": 0.0075952449114993215, "clip_ratio/region_mean": 0.015095244743861258, "completions/clipped_ratio": 0.0, "completions/max_length": 200.0, "completions/max_terminated_length": 200.0, "completions/mean_length": 198.75, "completions/mean_terminated_length": 198.75, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "entropy": 0.12045034673064947, "epoch": 0.1131863276700004, "frac_reward_zero_std": 0.0, "grad_norm": 1.7950763702392578, "learning_rate": 1.4636363636363637e-06, "loss": -0.002, "num_tokens": 6399800.0, "reward": 0.9085431098937988, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993983507156372, "reward_meter_std": 7.34872737666592e-05, "reward_repeat_penalty_mean": 0.9090909361839294, "reward_repeat_penalty_std": 0.048592954874038696, "reward_std": 0.04854761064052582, "reward_total_composite_mean": 0.9085431098937988, "reward_total_composite_std": 0.04854762554168701, "reward_total_mean": 0.9085431098937988, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993983507156372, "rewards/meter/std": 7.34872737666592e-05, "rewards/repeat_penalty/mean": 0.9090909361839294, "rewards/repeat_penalty/std": 0.048592954874038696, "rewards/total_composite/mean": 0.9085431098937988, "rewards/total_composite/std": 0.04854762554168701, "sampling/importance_sampling_ratio/max": 1.756600022315979, "sampling/importance_sampling_ratio/mean": 1.003812551498413, "sampling/importance_sampling_ratio/min": 0.3404516279697418, "sampling/sampling_logp_difference/max": 1.0774822235107422, "sampling/sampling_logp_difference/mean": 0.014119832776486874, "step": 2818 }, { "clip_ratio/high_max": 0.007020366610959172, "clip_ratio/high_mean": 0.007020366610959172, "clip_ratio/low_mean": 0.003537735901772976, "clip_ratio/low_min": 0.003537735901772976, "clip_ratio/region_mean": 0.010558102512732148, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 106.75, "completions/mean_terminated_length": 106.75, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.12008912675082684, "epoch": 0.11322649315178536, "frac_reward_zero_std": 0.0, "grad_norm": 4.404444217681885, "learning_rate": 1.4606060606060608e-06, "loss": -0.0006, "num_tokens": 6402006.0, "reward": 0.9992326498031616, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992326498031616, "reward_meter_std": 0.0003244501131121069, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00032444868702441454, "reward_total_composite_mean": 0.9992326498031616, "reward_total_composite_std": 0.0003244501131121069, "reward_total_mean": 0.9992326498031616, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992326498031616, "rewards/meter/std": 0.0003244501131121069, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992326498031616, "rewards/total_composite/std": 0.0003244501131121069, "sampling/importance_sampling_ratio/max": 1.5127817392349243, "sampling/importance_sampling_ratio/mean": 1.0044901371002197, "sampling/importance_sampling_ratio/min": 0.1967354267835617, "sampling/sampling_logp_difference/max": 1.6258955001831055, "sampling/sampling_logp_difference/mean": 0.01700535975396633, "step": 2819 }, { "clip_ratio/high_max": 0.00890342053025961, "clip_ratio/high_mean": 0.00890342053025961, "clip_ratio/low_mean": 0.008064515888690948, "clip_ratio/low_min": 0.008064515888690948, "clip_ratio/region_mean": 0.016967936418950558, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 69.625, "completions/mean_terminated_length": 69.625, "completions/min_length": 62.0, "completions/min_terminated_length": 62.0, "entropy": 0.10764285689219832, "epoch": 0.11326665863357031, "frac_reward_zero_std": 0.0, "grad_norm": 4.645050525665283, "learning_rate": 1.4575757575757576e-06, "loss": -0.0417, "num_tokens": 6403787.0, "reward": 0.9988994598388672, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988994598388672, "reward_meter_std": 0.001365774660371244, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0013657818781211972, "reward_total_composite_mean": 0.9988994598388672, "reward_total_composite_std": 0.001365774660371244, "reward_total_mean": 0.9988994598388672, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988994598388672, "rewards/meter/std": 0.001365774660371244, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988994598388672, "rewards/total_composite/std": 0.001365774660371244, "sampling/importance_sampling_ratio/max": 1.4191588163375854, "sampling/importance_sampling_ratio/mean": 0.9964150786399841, "sampling/importance_sampling_ratio/min": 0.1399795413017273, "sampling/sampling_logp_difference/max": 1.9662590026855469, "sampling/sampling_logp_difference/mean": 0.022290274500846863, "step": 2820 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0020125040755374357, "epoch": 0.11330682411535527, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.4545454545454546e-06, "loss": 0.0, "num_tokens": 6405779.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0028430223464966, "sampling/importance_sampling_ratio/mean": 1.0002089738845825, "sampling/importance_sampling_ratio/min": 0.9997273683547974, "sampling/sampling_logp_difference/max": 0.002838927786797285, "sampling/sampling_logp_difference/mean": 0.00021103888866491616, "step": 2821 }, { "clip_ratio/high_max": 0.003930817823857069, "clip_ratio/high_mean": 0.003930817823857069, "clip_ratio/low_mean": 0.003149755357299, "clip_ratio/low_min": 0.003149755357299, "clip_ratio/region_mean": 0.007080573181156069, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 158.625, "completions/mean_terminated_length": 158.625, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "entropy": 0.14032841473817825, "epoch": 0.11334698959714022, "frac_reward_zero_std": 0.0, "grad_norm": 1.043582558631897, "learning_rate": 1.4515151515151515e-06, "loss": -0.002, "num_tokens": 6408280.0, "reward": 0.9252284169197083, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.980668306350708, "reward_meter_std": 0.04880243167281151, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.0608980655670166, "reward_total_composite_mean": 0.9252284169197083, "reward_total_composite_std": 0.0608980655670166, "reward_total_mean": 0.9252284169197083, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.980668306350708, "rewards/meter/std": 0.04880243167281151, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.9252284169197083, "rewards/total_composite/std": 0.0608980655670166, "sampling/importance_sampling_ratio/max": 1.4671419858932495, "sampling/importance_sampling_ratio/mean": 1.0040552616119385, "sampling/importance_sampling_ratio/min": 0.32005739212036133, "sampling/sampling_logp_difference/max": 1.1392550468444824, "sampling/sampling_logp_difference/mean": 0.013364696875214577, "step": 2822 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.001565092708915472, "epoch": 0.11338715507892518, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.4484848484848485e-06, "loss": 0.0, "num_tokens": 6409984.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0021015405654907, "sampling/importance_sampling_ratio/mean": 1.000205397605896, "sampling/importance_sampling_ratio/min": 0.9997925162315369, "sampling/sampling_logp_difference/max": 0.0020993247162550688, "sampling/sampling_logp_difference/mean": 0.0002062379935523495, "step": 2823 }, { "clip_ratio/high_max": 0.0017361111240461469, "clip_ratio/high_mean": 0.0017361111240461469, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/region_mean": 0.0034722222480922937, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.06790398666635156, "epoch": 0.11342732056071013, "frac_reward_zero_std": 0.0, "grad_norm": 0.7708361148834229, "learning_rate": 1.4454545454545453e-06, "loss": 0.0014, "num_tokens": 6411779.0, "reward": 0.9994142055511475, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994142055511475, "reward_meter_std": 5.688391684088856e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.688605597242713e-05, "reward_total_composite_mean": 0.9994142055511475, "reward_total_composite_std": 5.688391684088856e-05, "reward_total_mean": 0.9994142055511475, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994142055511475, "rewards/meter/std": 5.688391684088856e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994142055511475, "rewards/total_composite/std": 5.688391684088856e-05, "sampling/importance_sampling_ratio/max": 1.4751698970794678, "sampling/importance_sampling_ratio/mean": 1.0028364658355713, "sampling/importance_sampling_ratio/min": 0.5081238150596619, "sampling/sampling_logp_difference/max": 0.677030086517334, "sampling/sampling_logp_difference/mean": 0.009274118579924107, "step": 2824 }, { "clip_ratio/high_max": 0.009345794096589088, "clip_ratio/high_mean": 0.009345794096589088, "clip_ratio/low_mean": 0.0035493926843628287, "clip_ratio/low_min": 0.0035493926843628287, "clip_ratio/region_mean": 0.012895186780951917, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 106.375, "completions/mean_terminated_length": 106.375, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.12096875347197056, "epoch": 0.11346748604249508, "frac_reward_zero_std": 0.0, "grad_norm": 0.8596204519271851, "learning_rate": 1.4424242424242426e-06, "loss": -0.0007, "num_tokens": 6413966.0, "reward": 0.9992821216583252, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992821216583252, "reward_meter_std": 0.0001443348592147231, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000144336445373483, "reward_total_composite_mean": 0.9992821216583252, "reward_total_composite_std": 0.0001443348592147231, "reward_total_mean": 0.9992821216583252, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992821216583252, "rewards/meter/std": 0.0001443348592147231, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992821216583252, "rewards/total_composite/std": 0.0001443348592147231, "sampling/importance_sampling_ratio/max": 1.6910252571105957, "sampling/importance_sampling_ratio/mean": 1.001145839691162, "sampling/importance_sampling_ratio/min": 0.2722080945968628, "sampling/sampling_logp_difference/max": 1.3011884689331055, "sampling/sampling_logp_difference/mean": 0.018452905118465424, "step": 2825 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.0037313431967049837, "clip_ratio/low_min": 0.0037313431967049837, "clip_ratio/region_mean": 0.005597014795057476, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.03601772431284189, "epoch": 0.11350765152428004, "frac_reward_zero_std": 0.0, "grad_norm": 0.9645191431045532, "learning_rate": 1.4393939393939396e-06, "loss": -0.0004, "num_tokens": 6415713.0, "reward": 0.9981287717819214, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981287717819214, "reward_meter_std": 3.6517179978545755e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.651110455393791e-05, "reward_total_composite_mean": 0.9981287717819214, "reward_total_composite_std": 3.6517179978545755e-05, "reward_total_mean": 0.9981287717819214, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981287717819214, "rewards/meter/std": 3.6517179978545755e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981287717819214, "rewards/total_composite/std": 3.6517179978545755e-05, "sampling/importance_sampling_ratio/max": 1.138381838798523, "sampling/importance_sampling_ratio/mean": 0.9988413453102112, "sampling/importance_sampling_ratio/min": 0.37584227323532104, "sampling/sampling_logp_difference/max": 0.9785857200622559, "sampling/sampling_logp_difference/mean": 0.007696997374296188, "step": 2826 }, { "clip_ratio/high_max": 0.008849852601997554, "clip_ratio/high_mean": 0.008849852601997554, "clip_ratio/low_mean": 0.005142431997228414, "clip_ratio/low_min": 0.005142431997228414, "clip_ratio/region_mean": 0.013992284599225968, "completions/clipped_ratio": 0.0, "completions/max_length": 201.0, "completions/max_terminated_length": 201.0, "completions/mean_length": 197.25, "completions/mean_terminated_length": 197.25, "completions/min_length": 192.0, "completions/min_terminated_length": 192.0, "entropy": 0.117509332485497, "epoch": 0.11354781700606499, "frac_reward_zero_std": 0.0, "grad_norm": 2.085141897201538, "learning_rate": 1.4363636363636365e-06, "loss": -0.0018, "num_tokens": 6418795.0, "reward": 0.9539901614189148, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994193315505981, "reward_meter_std": 5.9720292483689263e-05, "reward_repeat_penalty_mean": 0.9545454978942871, "reward_repeat_penalty_std": 0.0485929399728775, "reward_std": 0.04854191467165947, "reward_total_composite_mean": 0.9539901614189148, "reward_total_composite_std": 0.04854191467165947, "reward_total_mean": 0.9539901614189148, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994193315505981, "rewards/meter/std": 5.9720292483689263e-05, "rewards/repeat_penalty/mean": 0.9545454978942871, "rewards/repeat_penalty/std": 0.0485929399728775, "rewards/total_composite/mean": 0.9539901614189148, "rewards/total_composite/std": 0.04854191467165947, "sampling/importance_sampling_ratio/max": 1.5125662088394165, "sampling/importance_sampling_ratio/mean": 1.0017954111099243, "sampling/importance_sampling_ratio/min": 0.2597096562385559, "sampling/sampling_logp_difference/max": 1.3481910228729248, "sampling/sampling_logp_difference/mean": 0.015652773901820183, "step": 2827 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0012116766738472506, "epoch": 0.11358798248784995, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.4333333333333335e-06, "loss": 0.0, "num_tokens": 6420507.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0083885192871094, "sampling/importance_sampling_ratio/mean": 1.0001118183135986, "sampling/importance_sampling_ratio/min": 0.9991452097892761, "sampling/sampling_logp_difference/max": 0.008353465236723423, "sampling/sampling_logp_difference/mean": 0.00012123679334763438, "step": 2828 }, { "clip_ratio/high_max": 0.0013297871919348836, "clip_ratio/high_mean": 0.0013297871919348836, "clip_ratio/low_mean": 0.014784946106374264, "clip_ratio/low_min": 0.014784946106374264, "clip_ratio/region_mean": 0.016114733298309147, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 93.125, "completions/mean_terminated_length": 93.125, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.10609597992151976, "epoch": 0.1136281479696349, "frac_reward_zero_std": 0.0, "grad_norm": 1.4268546104431152, "learning_rate": 1.4303030303030306e-06, "loss": 0.0003, "num_tokens": 6422684.0, "reward": 0.9977059364318848, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977059364318848, "reward_meter_std": 6.68626744300127e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.687968561891466e-05, "reward_total_composite_mean": 0.9977059364318848, "reward_total_composite_std": 6.68626744300127e-05, "reward_total_mean": 0.9977059364318848, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977059364318848, "rewards/meter/std": 6.68626744300127e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977059364318848, "rewards/total_composite/std": 6.68626744300127e-05, "sampling/importance_sampling_ratio/max": 1.625991702079773, "sampling/importance_sampling_ratio/mean": 1.0083869695663452, "sampling/importance_sampling_ratio/min": 0.3639715313911438, "sampling/sampling_logp_difference/max": 1.0106797218322754, "sampling/sampling_logp_difference/mean": 0.0164635069668293, "step": 2829 }, { "clip_ratio/high_max": 0.010205571772530675, "clip_ratio/high_mean": 0.010205571772530675, "clip_ratio/low_mean": 0.004751874483190477, "clip_ratio/low_min": 0.004751874483190477, "clip_ratio/region_mean": 0.014957446255721152, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 158.375, "completions/mean_terminated_length": 158.375, "completions/min_length": 157.0, "completions/min_terminated_length": 157.0, "entropy": 0.16988196410238743, "epoch": 0.11366831345141985, "frac_reward_zero_std": 0.0, "grad_norm": 2.387721538543701, "learning_rate": 1.4272727272727274e-06, "loss": -0.0018, "num_tokens": 6425415.0, "reward": 0.9561665058135986, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977245330810547, "reward_meter_std": 0.0005517599638551474, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.05763239413499832, "reward_total_composite_mean": 0.9561665058135986, "reward_total_composite_std": 0.05763240531086922, "reward_total_mean": 0.9561665058135986, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977245330810547, "rewards/meter/std": 0.0005517599638551474, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9561665058135986, "rewards/total_composite/std": 0.05763240531086922, "sampling/importance_sampling_ratio/max": 1.5251771211624146, "sampling/importance_sampling_ratio/mean": 1.0021953582763672, "sampling/importance_sampling_ratio/min": 0.21695472300052643, "sampling/sampling_logp_difference/max": 1.528066635131836, "sampling/sampling_logp_difference/mean": 0.01837637834250927, "step": 2830 }, { "clip_ratio/high_max": 0.008177978219464421, "clip_ratio/high_mean": 0.008177978219464421, "clip_ratio/low_mean": 0.0072115384973585606, "clip_ratio/low_min": 0.0072115384973585606, "clip_ratio/region_mean": 0.015389516716822982, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 106.375, "completions/mean_terminated_length": 106.375, "completions/min_length": 104.0, "completions/min_terminated_length": 104.0, "entropy": 0.1615914311259985, "epoch": 0.1137084789332048, "frac_reward_zero_std": 0.0, "grad_norm": 3.8205645084381104, "learning_rate": 1.4242424242424244e-06, "loss": -0.0056, "num_tokens": 6427666.0, "reward": 0.8774728775024414, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8774728775024414, "reward_meter_std": 0.3444782793521881, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3444782495498657, "reward_total_composite_mean": 0.8774728775024414, "reward_total_composite_std": 0.3444782793521881, "reward_total_mean": 0.8774728775024414, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8774728775024414, "rewards/meter/std": 0.3444782793521881, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8774728775024414, "rewards/total_composite/std": 0.3444782793521881, "sampling/importance_sampling_ratio/max": 1.4661322832107544, "sampling/importance_sampling_ratio/mean": 1.0037299394607544, "sampling/importance_sampling_ratio/min": 0.37599948048591614, "sampling/sampling_logp_difference/max": 0.9781675338745117, "sampling/sampling_logp_difference/mean": 0.01733158528804779, "step": 2831 }, { "clip_ratio/high_max": 0.0015822785208001733, "clip_ratio/high_mean": 0.0015822785208001733, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0015822785208001733, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 79.875, "completions/mean_terminated_length": 79.875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.007047486782539636, "epoch": 0.11374864441498976, "frac_reward_zero_std": 0.0, "grad_norm": 0.9516367316246033, "learning_rate": 1.4212121212121213e-06, "loss": 0.0047, "num_tokens": 6429513.0, "reward": 0.7240880727767944, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7240880727767944, "reward_meter_std": 0.014326409436762333, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.014326424337923527, "reward_total_composite_mean": 0.7240880727767944, "reward_total_composite_std": 0.014326409436762333, "reward_total_mean": 0.7240880727767944, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7240880727767944, "rewards/meter/std": 0.014326409436762333, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7240880727767944, "rewards/total_composite/std": 0.014326409436762333, "sampling/importance_sampling_ratio/max": 1.1789082288742065, "sampling/importance_sampling_ratio/mean": 0.9994467496871948, "sampling/importance_sampling_ratio/min": 0.3448401391506195, "sampling/sampling_logp_difference/max": 1.0646743774414062, "sampling/sampling_logp_difference/mean": 0.002547384472563863, "step": 2832 }, { "clip_ratio/high_max": 0.010135134682059288, "clip_ratio/high_mean": 0.010135134682059288, "clip_ratio/low_mean": 0.003289473708719015, "clip_ratio/low_min": 0.003289473708719015, "clip_ratio/region_mean": 0.013424608390778303, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 35.25, "completions/mean_terminated_length": 35.25, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.12462522741407156, "epoch": 0.11378880989677471, "frac_reward_zero_std": 0.0, "grad_norm": 4.6605939865112305, "learning_rate": 1.4181818181818183e-06, "loss": 0.0162, "num_tokens": 6430923.0, "reward": 0.9975078105926514, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975078105926514, "reward_meter_std": 0.0002991736400872469, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00029916909988969564, "reward_total_composite_mean": 0.9975078105926514, "reward_total_composite_std": 0.0002991736400872469, "reward_total_mean": 0.9975078105926514, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975078105926514, "rewards/meter/std": 0.0002991736400872469, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975078105926514, "rewards/total_composite/std": 0.0002991736400872469, "sampling/importance_sampling_ratio/max": 1.2360522747039795, "sampling/importance_sampling_ratio/mean": 0.9931513667106628, "sampling/importance_sampling_ratio/min": 0.34551531076431274, "sampling/sampling_logp_difference/max": 1.062718391418457, "sampling/sampling_logp_difference/mean": 0.023440107703208923, "step": 2833 }, { "clip_ratio/high_max": 0.010262453462928534, "clip_ratio/high_mean": 0.010262453462928534, "clip_ratio/low_mean": 0.0023785202065482736, "clip_ratio/low_min": 0.0023785202065482736, "clip_ratio/region_mean": 0.012640973669476807, "completions/clipped_ratio": 0.0, "completions/max_length": 161.0, "completions/max_terminated_length": 161.0, "completions/mean_length": 158.125, "completions/mean_terminated_length": 158.125, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.19695778377354145, "epoch": 0.11382897537855967, "frac_reward_zero_std": 0.0, "grad_norm": 1.4452556371688843, "learning_rate": 1.4151515151515151e-06, "loss": 0.0036, "num_tokens": 6433732.0, "reward": 0.9522948265075684, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9938750267028809, "reward_meter_std": 0.01016020867973566, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.054876767098903656, "reward_total_composite_mean": 0.9522948265075684, "reward_total_composite_std": 0.05487677827477455, "reward_total_mean": 0.9522948265075684, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9938750267028809, "rewards/meter/std": 0.01016020867973566, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9522948265075684, "rewards/total_composite/std": 0.05487677827477455, "sampling/importance_sampling_ratio/max": 1.7436052560806274, "sampling/importance_sampling_ratio/mean": 1.001399040222168, "sampling/importance_sampling_ratio/min": 0.2555873394012451, "sampling/sampling_logp_difference/max": 1.3641910552978516, "sampling/sampling_logp_difference/mean": 0.021103786304593086, "step": 2834 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0018191997078247368, "epoch": 0.11386914086034462, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.4121212121212122e-06, "loss": 0.0, "num_tokens": 6435500.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0096708536148071, "sampling/importance_sampling_ratio/mean": 1.0001351833343506, "sampling/importance_sampling_ratio/min": 0.9914776682853699, "sampling/sampling_logp_difference/max": 0.009624381549656391, "sampling/sampling_logp_difference/mean": 0.00019080200581811368, "step": 2835 }, { "clip_ratio/high_max": 0.014033293817192316, "clip_ratio/high_mean": 0.014033293817192316, "clip_ratio/low_mean": 0.006538120680488646, "clip_ratio/low_min": 0.006538120680488646, "clip_ratio/region_mean": 0.020571414497680962, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 97.125, "completions/mean_terminated_length": 97.125, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.4101872891187668, "epoch": 0.11390930634212958, "frac_reward_zero_std": 0.0, "grad_norm": 3.5270473957061768, "learning_rate": 1.409090909090909e-06, "loss": -0.0092, "num_tokens": 6437557.0, "reward": 0.9715797901153564, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9715797901153564, "reward_meter_std": 0.03168648108839989, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03168648108839989, "reward_total_composite_mean": 0.9715797901153564, "reward_total_composite_std": 0.03168648108839989, "reward_total_mean": 0.9715797901153564, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9715797901153564, "rewards/meter/std": 0.03168648108839989, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9715797901153564, "rewards/total_composite/std": 0.03168648108839989, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0144469738006592, "sampling/importance_sampling_ratio/min": 0.3264382481575012, "sampling/sampling_logp_difference/max": 1.1195144653320312, "sampling/sampling_logp_difference/mean": 0.047103382647037506, "step": 2836 }, { "clip_ratio/high_max": 0.02511692512780428, "clip_ratio/high_mean": 0.02511692512780428, "clip_ratio/low_mean": 0.0031847134232521057, "clip_ratio/low_min": 0.0031847134232521057, "clip_ratio/region_mean": 0.028301638551056385, "completions/clipped_ratio": 0.0, "completions/max_length": 158.0, "completions/max_terminated_length": 158.0, "completions/mean_length": 155.625, "completions/mean_terminated_length": 155.625, "completions/min_length": 151.0, "completions/min_terminated_length": 151.0, "entropy": 0.23529152572155, "epoch": 0.11394947182391453, "frac_reward_zero_std": 0.0, "grad_norm": 2.079005002975464, "learning_rate": 1.406060606060606e-06, "loss": 0.0116, "num_tokens": 6440330.0, "reward": 0.9696353077888489, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973591566085815, "reward_meter_std": 0.001539723016321659, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05092417448759079, "reward_total_composite_mean": 0.9696353077888489, "reward_total_composite_std": 0.05092417448759079, "reward_total_mean": 0.9696353077888489, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973591566085815, "rewards/meter/std": 0.001539723016321659, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9696353077888489, "rewards/total_composite/std": 0.05092417448759079, "sampling/importance_sampling_ratio/max": 1.7507938146591187, "sampling/importance_sampling_ratio/mean": 1.0057966709136963, "sampling/importance_sampling_ratio/min": 0.1186932623386383, "sampling/sampling_logp_difference/max": 2.1312127113342285, "sampling/sampling_logp_difference/mean": 0.0309575367718935, "step": 2837 }, { "clip_ratio/high_max": 0.012895984342321754, "clip_ratio/high_mean": 0.012895984342321754, "clip_ratio/low_mean": 0.023296055383980274, "clip_ratio/low_min": 0.023296055383980274, "clip_ratio/region_mean": 0.03619203972630203, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 134.5, "completions/mean_terminated_length": 134.5, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.37597112730145454, "epoch": 0.11398963730569948, "frac_reward_zero_std": 0.0, "grad_norm": 2.711418628692627, "learning_rate": 1.403030303030303e-06, "loss": -0.0031, "num_tokens": 6442782.0, "reward": 0.999076247215271, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999076247215271, "reward_meter_std": 0.00021775417553726584, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00021775579079985619, "reward_total_composite_mean": 0.999076247215271, "reward_total_composite_std": 0.00021775417553726584, "reward_total_mean": 0.999076247215271, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999076247215271, "rewards/meter/std": 0.00021775417553726584, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999076247215271, "rewards/total_composite/std": 0.00021775417553726584, "sampling/importance_sampling_ratio/max": 1.565982460975647, "sampling/importance_sampling_ratio/mean": 1.0005239248275757, "sampling/importance_sampling_ratio/min": 0.26113519072532654, "sampling/sampling_logp_difference/max": 1.342717170715332, "sampling/sampling_logp_difference/mean": 0.044978611171245575, "step": 2838 }, { "clip_ratio/high_max": 0.0361340157687664, "clip_ratio/high_mean": 0.0361340157687664, "clip_ratio/low_mean": 0.007575757801532745, "clip_ratio/low_min": 0.007575757801532745, "clip_ratio/region_mean": 0.04370977357029915, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.19998472929000854, "epoch": 0.11402980278748444, "frac_reward_zero_std": 0.0, "grad_norm": 4.530300140380859, "learning_rate": 1.4000000000000001e-06, "loss": 0.0133, "num_tokens": 6444599.0, "reward": 0.9253357648849487, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9489889144897461, "reward_meter_std": 0.003239382989704609, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06812427937984467, "reward_total_composite_mean": 0.9253357648849487, "reward_total_composite_std": 0.06812429428100586, "reward_total_mean": 0.9253357648849487, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9489889144897461, "rewards/meter/std": 0.003239382989704609, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9253357648849487, "rewards/total_composite/std": 0.06812429428100586, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0046266317367554, "sampling/importance_sampling_ratio/min": 0.2307029664516449, "sampling/sampling_logp_difference/max": 1.4666242599487305, "sampling/sampling_logp_difference/mean": 0.039003822952508926, "step": 2839 }, { "clip_ratio/high_max": 0.005076272878795862, "clip_ratio/high_mean": 0.005076272878795862, "clip_ratio/low_mean": 0.0038265305338427424, "clip_ratio/low_min": 0.0038265305338427424, "clip_ratio/region_mean": 0.008902803412638605, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 98.125, "completions/mean_terminated_length": 98.125, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.04481238219887018, "epoch": 0.11406996826926939, "frac_reward_zero_std": 0.0, "grad_norm": 0.9888406991958618, "learning_rate": 1.3969696969696972e-06, "loss": 0.001, "num_tokens": 6446736.0, "reward": 0.9993867874145508, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993867874145508, "reward_meter_std": 5.2988518291385844e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.3000439947936684e-05, "reward_total_composite_mean": 0.9993867874145508, "reward_total_composite_std": 5.2988518291385844e-05, "reward_total_mean": 0.9993867874145508, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993867874145508, "rewards/meter/std": 5.2988518291385844e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993867874145508, "rewards/total_composite/std": 5.2988518291385844e-05, "sampling/importance_sampling_ratio/max": 1.6428247690200806, "sampling/importance_sampling_ratio/mean": 1.0002104043960571, "sampling/importance_sampling_ratio/min": 0.326957643032074, "sampling/sampling_logp_difference/max": 1.117924690246582, "sampling/sampling_logp_difference/mean": 0.009926079772412777, "step": 2840 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.001806725820642896, "epoch": 0.11411013375105435, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.3939393939393942e-06, "loss": 0.0, "num_tokens": 6448576.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0026506185531616, "sampling/importance_sampling_ratio/mean": 1.000229001045227, "sampling/importance_sampling_ratio/min": 0.9991210103034973, "sampling/sampling_logp_difference/max": 0.0026471014134585857, "sampling/sampling_logp_difference/mean": 0.00023380780476145446, "step": 2841 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.002088540146360174, "epoch": 0.1141502992328393, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.390909090909091e-06, "loss": 0.0, "num_tokens": 6450400.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0035706758499146, "sampling/importance_sampling_ratio/mean": 1.0002295970916748, "sampling/importance_sampling_ratio/min": 0.9999062418937683, "sampling/sampling_logp_difference/max": 0.003564245533198118, "sampling/sampling_logp_difference/mean": 0.00023029858130030334, "step": 2842 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0019821640598820522, "epoch": 0.11419046471462425, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.3878787878787881e-06, "loss": 0.0, "num_tokens": 6452232.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0022305250167847, "sampling/importance_sampling_ratio/mean": 1.0002238750457764, "sampling/importance_sampling_ratio/min": 0.999897837638855, "sampling/sampling_logp_difference/max": 0.0022279510740190744, "sampling/sampling_logp_difference/mean": 0.0002246771182399243, "step": 2843 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0018740687082754448, "epoch": 0.11423063019640921, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.384848484848485e-06, "loss": 0.0, "num_tokens": 6453936.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0034716129302979, "sampling/importance_sampling_ratio/mean": 1.0002586841583252, "sampling/importance_sampling_ratio/min": 0.9998733997344971, "sampling/sampling_logp_difference/max": 0.003465580753982067, "sampling/sampling_logp_difference/mean": 0.00025937738246284425, "step": 2844 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.004738863033708185, "epoch": 0.11427079567819416, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.381818181818182e-06, "loss": 0.0, "num_tokens": 6455448.0, "reward": 0.9980231523513794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9980231523513794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.008884072303772, "sampling/importance_sampling_ratio/mean": 1.000425934791565, "sampling/importance_sampling_ratio/min": 0.9850959777832031, "sampling/sampling_logp_difference/max": 0.01501617580652237, "sampling/sampling_logp_difference/mean": 0.0005941917770542204, "step": 2845 }, { "clip_ratio/high_max": 0.008395581040531397, "clip_ratio/high_mean": 0.008395581040531397, "clip_ratio/low_mean": 0.007716872845776379, "clip_ratio/low_min": 0.007716872845776379, "clip_ratio/region_mean": 0.016112453886307776, "completions/clipped_ratio": 0.0, "completions/max_length": 179.0, "completions/max_terminated_length": 179.0, "completions/mean_length": 178.375, "completions/mean_terminated_length": 178.375, "completions/min_length": 178.0, "completions/min_terminated_length": 178.0, "entropy": 0.18307929299771786, "epoch": 0.11431096115997912, "frac_reward_zero_std": 0.0, "grad_norm": 2.306335926055908, "learning_rate": 1.3787878787878788e-06, "loss": 0.0054, "num_tokens": 6458571.0, "reward": 0.9434767961502075, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989786148071289, "reward_meter_std": 0.00010886832023970783, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.05927610024809837, "reward_total_composite_mean": 0.9434767961502075, "reward_total_composite_std": 0.05927610024809837, "reward_total_mean": 0.9434767961502075, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989786148071289, "rewards/meter/std": 0.00010886832023970783, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.9434767961502075, "rewards/total_composite/std": 0.05927610024809837, "sampling/importance_sampling_ratio/max": 1.79298734664917, "sampling/importance_sampling_ratio/mean": 1.0038070678710938, "sampling/importance_sampling_ratio/min": 0.36631348729133606, "sampling/sampling_logp_difference/max": 1.0042657852172852, "sampling/sampling_logp_difference/mean": 0.01770571805536747, "step": 2846 }, { "clip_ratio/high_max": 0.016998398234136403, "clip_ratio/high_mean": 0.016998398234136403, "clip_ratio/low_mean": 0.01566602219827473, "clip_ratio/low_min": 0.01566602219827473, "clip_ratio/region_mean": 0.032664420432411134, "completions/clipped_ratio": 0.0, "completions/max_length": 82.0, "completions/max_terminated_length": 82.0, "completions/mean_length": 80.375, "completions/mean_terminated_length": 80.375, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.2891568746417761, "epoch": 0.11435112664176407, "frac_reward_zero_std": 0.0, "grad_norm": 2.3205626010894775, "learning_rate": 1.3757575757575759e-06, "loss": -0.0027, "num_tokens": 6460630.0, "reward": 0.9975742101669312, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975742101669312, "reward_meter_std": 0.002095119096338749, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0020951295737177134, "reward_total_composite_mean": 0.9975742101669312, "reward_total_composite_std": 0.002095119096338749, "reward_total_mean": 0.9975742101669312, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975742101669312, "rewards/meter/std": 0.002095119096338749, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975742101669312, "rewards/total_composite/std": 0.002095119096338749, "sampling/importance_sampling_ratio/max": 1.7717058658599854, "sampling/importance_sampling_ratio/mean": 1.0142433643341064, "sampling/importance_sampling_ratio/min": 0.3252760171890259, "sampling/sampling_logp_difference/max": 1.1230812072753906, "sampling/sampling_logp_difference/mean": 0.034449756145477295, "step": 2847 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.005831975577166304, "epoch": 0.11439129212354902, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.3727272727272727e-06, "loss": 0.0, "num_tokens": 6462110.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.017333745956421, "sampling/importance_sampling_ratio/mean": 1.0002989768981934, "sampling/importance_sampling_ratio/min": 0.9707706570625305, "sampling/sampling_logp_difference/max": 0.029664982110261917, "sampling/sampling_logp_difference/mean": 0.0006461237207986414, "step": 2848 }, { "clip_ratio/high_max": 0.0078856002073735, "clip_ratio/high_mean": 0.0078856002073735, "clip_ratio/low_mean": 0.005313260713592172, "clip_ratio/low_min": 0.005313260713592172, "clip_ratio/region_mean": 0.013198860920965672, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 142.25, "completions/mean_terminated_length": 142.25, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.15198146365582943, "epoch": 0.11443145760533398, "frac_reward_zero_std": 0.0, "grad_norm": 2.436666965484619, "learning_rate": 1.3696969696969697e-06, "loss": -0.0004, "num_tokens": 6464768.0, "reward": 0.8920784592628479, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991288781166077, "reward_meter_std": 0.0001261577708646655, "reward_repeat_penalty_mean": 0.8928571939468384, "reward_repeat_penalty_std": 0.10101524740457535, "reward_std": 0.10092535614967346, "reward_total_composite_mean": 0.8920784592628479, "reward_total_composite_std": 0.10092534869909286, "reward_total_mean": 0.8920784592628479, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991288781166077, "rewards/meter/std": 0.0001261577708646655, "rewards/repeat_penalty/mean": 0.8928571939468384, "rewards/repeat_penalty/std": 0.10101524740457535, "rewards/total_composite/mean": 0.8920784592628479, "rewards/total_composite/std": 0.10092534869909286, "sampling/importance_sampling_ratio/max": 1.7561588287353516, "sampling/importance_sampling_ratio/mean": 1.0004011392593384, "sampling/importance_sampling_ratio/min": 0.2321789264678955, "sampling/sampling_logp_difference/max": 1.4602470397949219, "sampling/sampling_logp_difference/mean": 0.019777992740273476, "step": 2849 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.002160783580620773, "epoch": 0.11447162308711893, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.3666666666666668e-06, "loss": 0.0, "num_tokens": 6466328.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0029124021530151, "sampling/importance_sampling_ratio/mean": 1.0002659559249878, "sampling/importance_sampling_ratio/min": 0.999262273311615, "sampling/sampling_logp_difference/max": 0.0029080850072205067, "sampling/sampling_logp_difference/mean": 0.0002724671212490648, "step": 2850 }, { "epoch": 0.11447162308711893, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/max_length": 411.38461538461536, "eval_completions/max_terminated_length": 386.0769230769231, "eval_completions/mean_length": 211.08653846153845, "eval_completions/mean_terminated_length": 201.30357360839844, "eval_completions/min_length": 59.92307692307692, "eval_completions/min_terminated_length": 59.92307692307692, "eval_entropy": 0.3772245920621432, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 6466328.0, "eval_reward": 0.6904892875598028, "eval_reward_arabic_clean_mean": 0.9615384615384616, "eval_reward_arabic_clean_std": 0.10878565678229699, "eval_reward_count_adherence_mean": 0.9523772459763747, "eval_reward_count_adherence_std": 0.06613255492769755, "eval_reward_meter_mean": 0.7710461295568026, "eval_reward_meter_std": 0.3627813183344327, "eval_reward_repeat_penalty_mean": 0.9531235878284161, "eval_reward_repeat_penalty_std": 0.07651243057961647, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.6904892875598028, "eval_reward_total_composite_std": 0.3804844663693355, "eval_reward_total_mean": 0.6904892875598028, "eval_rewards/arabic_clean/mean": 0.9615384615384616, "eval_rewards/arabic_clean/std": 0.10878565678229699, "eval_rewards/count_adherence/mean": 0.9523772459763747, "eval_rewards/count_adherence/std": 0.06613255492769755, "eval_rewards/meter/mean": 0.7710461295568026, "eval_rewards/meter/std": 0.3627813183344327, "eval_rewards/repeat_penalty/mean": 0.9531235878284161, "eval_rewards/repeat_penalty/std": 0.07651243057961647, "eval_rewards/total_composite/mean": 0.6904892875598028, "eval_rewards/total_composite/std": 0.3804844663693355, "eval_runtime": 77.18, "eval_samples_per_second": 1.347, "eval_sampling/importance_sampling_ratio/max": 1.5361342888612013, "eval_sampling/importance_sampling_ratio/mean": 1.009327219082759, "eval_sampling/importance_sampling_ratio/min": 0.33661178098275113, "eval_sampling/sampling_logp_difference/max": 1.119479619539701, "eval_sampling/sampling_logp_difference/mean": 0.03192363708065106, "eval_steps_per_second": 0.168, "step": 2850 }, { "clip_ratio/high_max": 0.010042233159765601, "clip_ratio/high_mean": 0.010042233159765601, "clip_ratio/low_mean": 0.004026185255497694, "clip_ratio/low_min": 0.004026185255497694, "clip_ratio/region_mean": 0.014068418415263295, "completions/clipped_ratio": 0.0, "completions/max_length": 250.0, "completions/max_terminated_length": 250.0, "completions/mean_length": 248.75, "completions/mean_terminated_length": 248.75, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.24152424931526184, "epoch": 0.11451178856890389, "frac_reward_zero_std": 0.0, "grad_norm": 1.7204653024673462, "learning_rate": 1.3636363636363636e-06, "loss": 0.0037, "num_tokens": 6469974.0, "reward": 0.9217178821563721, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985492825508118, "reward_meter_std": 0.000639773381408304, "reward_repeat_penalty_mean": 0.9230769276618958, "reward_repeat_penalty_std": 0.07121692597866058, "reward_std": 0.070813849568367, "reward_total_composite_mean": 0.9217178821563721, "reward_total_composite_std": 0.0708138644695282, "reward_total_mean": 0.9217178821563721, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985492825508118, "rewards/meter/std": 0.000639773381408304, "rewards/repeat_penalty/mean": 0.9230769276618958, "rewards/repeat_penalty/std": 0.07121692597866058, "rewards/total_composite/mean": 0.9217178821563721, "rewards/total_composite/std": 0.0708138644695282, "sampling/importance_sampling_ratio/max": 1.587019920349121, "sampling/importance_sampling_ratio/mean": 1.0041165351867676, "sampling/importance_sampling_ratio/min": 0.2292836457490921, "sampling/sampling_logp_difference/max": 1.4727954864501953, "sampling/sampling_logp_difference/mean": 0.024417459964752197, "step": 2851 }, { "clip_ratio/high_max": 0.014985310845077038, "clip_ratio/high_mean": 0.014985310845077038, "clip_ratio/low_mean": 0.008333333767950535, "clip_ratio/low_min": 0.008333333767950535, "clip_ratio/region_mean": 0.023318644613027573, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 65.5, "completions/mean_terminated_length": 65.5, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.38093627616763115, "epoch": 0.11455195405068884, "frac_reward_zero_std": 0.0, "grad_norm": 8.037017822265625, "learning_rate": 1.3606060606060607e-06, "loss": -0.0252, "num_tokens": 6471754.0, "reward": 0.951325535774231, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.951325535774231, "reward_meter_std": 0.1125522255897522, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.112552210688591, "reward_total_composite_mean": 0.951325535774231, "reward_total_composite_std": 0.1125522255897522, "reward_total_mean": 0.951325535774231, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.951325535774231, "rewards/meter/std": 0.1125522255897522, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.951325535774231, "rewards/total_composite/std": 0.1125522255897522, "sampling/importance_sampling_ratio/max": 1.7486579418182373, "sampling/importance_sampling_ratio/mean": 1.0115268230438232, "sampling/importance_sampling_ratio/min": 0.3231185972690582, "sampling/sampling_logp_difference/max": 1.1297359466552734, "sampling/sampling_logp_difference/mean": 0.042092662304639816, "step": 2852 }, { "clip_ratio/high_max": 0.0204123689327389, "clip_ratio/high_mean": 0.0204123689327389, "clip_ratio/low_mean": 0.004707278567366302, "clip_ratio/low_min": 0.004707278567366302, "clip_ratio/region_mean": 0.025119647500105202, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 79.875, "completions/mean_terminated_length": 79.875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.2539651710540056, "epoch": 0.1145921195324738, "frac_reward_zero_std": 0.0, "grad_norm": 2.253185510635376, "learning_rate": 1.357575757575758e-06, "loss": 0.0008, "num_tokens": 6473577.0, "reward": 0.9988980293273926, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988980293273926, "reward_meter_std": 0.0002794552710838616, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002794554748106748, "reward_total_composite_mean": 0.9988980293273926, "reward_total_composite_std": 0.0002794552710838616, "reward_total_mean": 0.9988980293273926, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988980293273926, "rewards/meter/std": 0.0002794552710838616, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988980293273926, "rewards/total_composite/std": 0.0002794552710838616, "sampling/importance_sampling_ratio/max": 1.6166248321533203, "sampling/importance_sampling_ratio/mean": 1.0052258968353271, "sampling/importance_sampling_ratio/min": 0.38139283657073975, "sampling/sampling_logp_difference/max": 0.9639253616333008, "sampling/sampling_logp_difference/mean": 0.027408134192228317, "step": 2853 }, { "clip_ratio/high_max": 0.018657967331819236, "clip_ratio/high_mean": 0.018657967331819236, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/region_mean": 0.02269022527616471, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 126.375, "completions/mean_terminated_length": 126.375, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.34479158371686935, "epoch": 0.11463228501425875, "frac_reward_zero_std": 0.0, "grad_norm": 4.1874613761901855, "learning_rate": 1.3545454545454547e-06, "loss": -0.0055, "num_tokens": 6475980.0, "reward": 0.929327130317688, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9470775127410889, "reward_meter_std": 0.12438784539699554, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.12683993577957153, "reward_total_composite_mean": 0.929327130317688, "reward_total_composite_std": 0.12683995068073273, "reward_total_mean": 0.929327130317688, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9470775127410889, "rewards/meter/std": 0.12438784539699554, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.929327130317688, "rewards/total_composite/std": 0.12683995068073273, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0064387321472168, "sampling/importance_sampling_ratio/min": 0.3374670445919037, "sampling/sampling_logp_difference/max": 1.086287498474121, "sampling/sampling_logp_difference/mean": 0.0373864509165287, "step": 2854 }, { "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.008196720853447914, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.06320370081812143, "epoch": 0.1146724504960437, "frac_reward_zero_std": 0.0, "grad_norm": 5.475809097290039, "learning_rate": 1.3515151515151518e-06, "loss": 0.0057, "num_tokens": 6477692.0, "reward": 0.9972702860832214, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972702860832214, "reward_meter_std": 0.00018694026221055537, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001869339175755158, "reward_total_composite_mean": 0.9972702860832214, "reward_total_composite_std": 0.00018694026221055537, "reward_total_mean": 0.9972702860832214, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972702860832214, "rewards/meter/std": 0.00018694026221055537, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972702860832214, "rewards/total_composite/std": 0.00018694026221055537, "sampling/importance_sampling_ratio/max": 1.7445498704910278, "sampling/importance_sampling_ratio/mean": 1.0026929378509521, "sampling/importance_sampling_ratio/min": 0.4347377419471741, "sampling/sampling_logp_difference/max": 0.8330123424530029, "sampling/sampling_logp_difference/mean": 0.01253997441381216, "step": 2855 }, { "clip_ratio/high_max": 0.02479799627326429, "clip_ratio/high_mean": 0.02479799627326429, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/region_mean": 0.02607350645121187, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 100.875, "completions/mean_terminated_length": 100.875, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.2935164403170347, "epoch": 0.11471261597782866, "frac_reward_zero_std": 0.0, "grad_norm": 3.800494909286499, "learning_rate": 1.3484848484848486e-06, "loss": -0.0073, "num_tokens": 6479747.0, "reward": 0.9850212931632996, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9850212931632996, "reward_meter_std": 0.02900601737201214, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02900601364672184, "reward_total_composite_mean": 0.9850212931632996, "reward_total_composite_std": 0.02900601737201214, "reward_total_mean": 0.9850212931632996, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9850212931632996, "rewards/meter/std": 0.02900601737201214, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9850212931632996, "rewards/total_composite/std": 0.02900601737201214, "sampling/importance_sampling_ratio/max": 1.8992688655853271, "sampling/importance_sampling_ratio/mean": 1.0072658061981201, "sampling/importance_sampling_ratio/min": 0.25652286410331726, "sampling/sampling_logp_difference/max": 1.3605375289916992, "sampling/sampling_logp_difference/mean": 0.031728874891996384, "step": 2856 }, { "clip_ratio/high_max": 0.009469697251915932, "clip_ratio/high_mean": 0.009469697251915932, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.009469697251915932, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 66.0, "completions/mean_terminated_length": 66.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.053996546659618616, "epoch": 0.11475278145961361, "frac_reward_zero_std": 0.0, "grad_norm": 0.22853942215442657, "learning_rate": 1.3454545454545457e-06, "loss": 0.0001, "num_tokens": 6481515.0, "reward": 0.9981458783149719, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981458783149719, "reward_meter_std": 1.3051687346887775e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.3051648238615599e-05, "reward_total_composite_mean": 0.9981458783149719, "reward_total_composite_std": 1.3051687346887775e-05, "reward_total_mean": 0.9981458783149719, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981458783149719, "rewards/meter/std": 1.3051687346887775e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981458783149719, "rewards/total_composite/std": 1.3051687346887775e-05, "sampling/importance_sampling_ratio/max": 1.469744324684143, "sampling/importance_sampling_ratio/mean": 1.002258539199829, "sampling/importance_sampling_ratio/min": 0.8009796142578125, "sampling/sampling_logp_difference/max": 0.3850884437561035, "sampling/sampling_logp_difference/mean": 0.005550025030970573, "step": 2857 }, { "clip_ratio/high_max": 0.0062042842619121075, "clip_ratio/high_mean": 0.0062042842619121075, "clip_ratio/low_mean": 0.014873409294523299, "clip_ratio/low_min": 0.014873409294523299, "clip_ratio/region_mean": 0.021077693556435406, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 279.875, "completions/mean_terminated_length": 279.875, "completions/min_length": 261.0, "completions/min_terminated_length": 261.0, "entropy": 0.2756977826356888, "epoch": 0.11479294694139856, "frac_reward_zero_std": 0.0, "grad_norm": 2.1501612663269043, "learning_rate": 1.3424242424242425e-06, "loss": 0.0307, "num_tokens": 6485258.0, "reward": 0.8241852521896362, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_meter_mean": 0.9832162857055664, "reward_meter_std": 0.0437212772667408, "reward_repeat_penalty_mean": 0.8965686559677124, "reward_repeat_penalty_std": 0.05159881338477135, "reward_std": 0.05549176037311554, "reward_total_composite_mean": 0.8241852521896362, "reward_total_composite_std": 0.05549175664782524, "reward_total_mean": 0.8241852521896362, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/meter/mean": 0.9832162857055664, "rewards/meter/std": 0.0437212772667408, "rewards/repeat_penalty/mean": 0.8965686559677124, "rewards/repeat_penalty/std": 0.05159881338477135, "rewards/total_composite/mean": 0.8241852521896362, "rewards/total_composite/std": 0.05549175664782524, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0033777952194214, "sampling/importance_sampling_ratio/min": 0.08637219667434692, "sampling/sampling_logp_difference/max": 2.449089527130127, "sampling/sampling_logp_difference/mean": 0.03513329103589058, "step": 2858 }, { "clip_ratio/high_max": 0.01487299520522356, "clip_ratio/high_mean": 0.01487299520522356, "clip_ratio/low_mean": 0.009471436496824026, "clip_ratio/low_min": 0.009471436496824026, "clip_ratio/region_mean": 0.024344431702047586, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.23347646743059158, "epoch": 0.11483311242318352, "frac_reward_zero_std": 0.0, "grad_norm": 3.014261245727539, "learning_rate": 1.3393939393939395e-06, "loss": -0.0101, "num_tokens": 6487078.0, "reward": 0.9922887086868286, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9922887086868286, "reward_meter_std": 0.0029313687700778246, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0029313701670616865, "reward_total_composite_mean": 0.9922887086868286, "reward_total_composite_std": 0.0029313687700778246, "reward_total_mean": 0.9922887086868286, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9922887086868286, "rewards/meter/std": 0.0029313687700778246, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9922887086868286, "rewards/total_composite/std": 0.0029313687700778246, "sampling/importance_sampling_ratio/max": 1.5588260889053345, "sampling/importance_sampling_ratio/mean": 1.0089040994644165, "sampling/importance_sampling_ratio/min": 0.3521854281425476, "sampling/sampling_logp_difference/max": 1.0435974597930908, "sampling/sampling_logp_difference/mean": 0.031141582876443863, "step": 2859 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.011363636702299118, "clip_ratio/low_min": 0.011363636702299118, "clip_ratio/region_mean": 0.011363636702299118, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.05899450369179249, "epoch": 0.11487327790496847, "frac_reward_zero_std": 0.0, "grad_norm": 0.3336952030658722, "learning_rate": 1.3363636363636364e-06, "loss": -0.0013, "num_tokens": 6488807.0, "reward": 0.9981107711791992, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981107711791992, "reward_meter_std": 5.631321982946247e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.631066233036108e-05, "reward_total_composite_mean": 0.9981107711791992, "reward_total_composite_std": 5.631321982946247e-05, "reward_total_mean": 0.9981107711791992, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981107711791992, "rewards/meter/std": 5.631321982946247e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981107711791992, "rewards/total_composite/std": 5.631321982946247e-05, "sampling/importance_sampling_ratio/max": 1.383058786392212, "sampling/importance_sampling_ratio/mean": 1.004895567893982, "sampling/importance_sampling_ratio/min": 0.6034249067306519, "sampling/sampling_logp_difference/max": 0.5051336288452148, "sampling/sampling_logp_difference/mean": 0.007993371225893497, "step": 2860 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0002626157729537226, "epoch": 0.11491344338675342, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.3333333333333334e-06, "loss": 0.0, "num_tokens": 6490287.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0006294250488281, "sampling/importance_sampling_ratio/mean": 1.0000324249267578, "sampling/importance_sampling_ratio/min": 0.9998278617858887, "sampling/sampling_logp_difference/max": 0.0006292310426943004, "sampling/sampling_logp_difference/mean": 3.3981952583417296e-05, "step": 2861 }, { "clip_ratio/high_max": 0.040064468048512936, "clip_ratio/high_mean": 0.040064468048512936, "clip_ratio/low_mean": 0.01037695212289691, "clip_ratio/low_min": 0.01037695212289691, "clip_ratio/region_mean": 0.050441420171409845, "completions/clipped_ratio": 0.0, "completions/max_length": 207.0, "completions/max_terminated_length": 207.0, "completions/mean_length": 195.0, "completions/mean_terminated_length": 195.0, "completions/min_length": 173.0, "completions/min_terminated_length": 173.0, "entropy": 0.4370429702103138, "epoch": 0.11495360886853838, "frac_reward_zero_std": 0.0, "grad_norm": 3.7277348041534424, "learning_rate": 1.3303030303030305e-06, "loss": -0.0358, "num_tokens": 6493447.0, "reward": 0.9573421478271484, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9583333134651184, "reward_count_adherence_std": 0.07715168595314026, "reward_meter_mean": 0.9989534616470337, "reward_meter_std": 0.00048265265650115907, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.07722420245409012, "reward_total_composite_mean": 0.9573421478271484, "reward_total_composite_std": 0.07722419500350952, "reward_total_mean": 0.9573421478271484, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9583333134651184, "rewards/count_adherence/std": 0.07715168595314026, "rewards/meter/mean": 0.9989534616470337, "rewards/meter/std": 0.00048265265650115907, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9573421478271484, "rewards/total_composite/std": 0.07722419500350952, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0075865983963013, "sampling/importance_sampling_ratio/min": 0.14445599913597107, "sampling/sampling_logp_difference/max": 1.9347803592681885, "sampling/sampling_logp_difference/mean": 0.05510193854570389, "step": 2862 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.005208333372138441, "clip_ratio/low_min": 0.005208333372138441, "clip_ratio/region_mean": 0.005208333372138441, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.07642534840852022, "epoch": 0.11499377435032333, "frac_reward_zero_std": 0.0, "grad_norm": 0.6136480569839478, "learning_rate": 1.3272727272727273e-06, "loss": 0.0, "num_tokens": 6495258.0, "reward": 0.9994188547134399, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994188547134399, "reward_meter_std": 3.3046217140508816e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.304608617327176e-05, "reward_total_composite_mean": 0.9994188547134399, "reward_total_composite_std": 3.3046217140508816e-05, "reward_total_mean": 0.9994188547134399, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994188547134399, "rewards/meter/std": 3.3046217140508816e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994188547134399, "rewards/total_composite/std": 3.3046217140508816e-05, "sampling/importance_sampling_ratio/max": 1.2199125289916992, "sampling/importance_sampling_ratio/mean": 1.0004888772964478, "sampling/importance_sampling_ratio/min": 0.3558204770088196, "sampling/sampling_logp_difference/max": 1.0333290100097656, "sampling/sampling_logp_difference/mean": 0.01137758232653141, "step": 2863 }, { "clip_ratio/high_max": 0.013674213085323572, "clip_ratio/high_mean": 0.013674213085323572, "clip_ratio/low_mean": 0.0030303029343485832, "clip_ratio/low_min": 0.0030303029343485832, "clip_ratio/region_mean": 0.016704516019672155, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 164.875, "completions/mean_terminated_length": 164.875, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.15149082150310278, "epoch": 0.11503393983210829, "frac_reward_zero_std": 0.0, "grad_norm": 2.3443167209625244, "learning_rate": 1.3242424242424243e-06, "loss": 0.0017, "num_tokens": 6498081.0, "reward": 0.9854899644851685, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993668794631958, "reward_meter_std": 0.0001238392578670755, "reward_repeat_penalty_mean": 0.9861111044883728, "reward_repeat_penalty_std": 0.03928370773792267, "reward_std": 0.039342816919088364, "reward_total_composite_mean": 0.9854899644851685, "reward_total_composite_std": 0.03934282064437866, "reward_total_mean": 0.9854899644851685, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993668794631958, "rewards/meter/std": 0.0001238392578670755, "rewards/repeat_penalty/mean": 0.9861111044883728, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.9854899644851685, "rewards/total_composite/std": 0.03934282064437866, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.000923991203308, "sampling/importance_sampling_ratio/min": 0.17922841012477875, "sampling/sampling_logp_difference/max": 1.7190942764282227, "sampling/sampling_logp_difference/mean": 0.024905560538172722, "step": 2864 }, { "clip_ratio/high_max": 0.019567076116800308, "clip_ratio/high_mean": 0.019567076116800308, "clip_ratio/low_mean": 0.008053920813836157, "clip_ratio/low_min": 0.008053920813836157, "clip_ratio/region_mean": 0.027620996930636466, "completions/clipped_ratio": 0.0, "completions/max_length": 470.0, "completions/max_terminated_length": 470.0, "completions/mean_length": 442.0, "completions/mean_terminated_length": 442.0, "completions/min_length": 430.0, "completions/min_terminated_length": 430.0, "entropy": 0.4746365137398243, "epoch": 0.11507410531389324, "frac_reward_zero_std": 0.0, "grad_norm": 1.4304178953170776, "learning_rate": 1.3212121212121212e-06, "loss": -0.0115, "num_tokens": 6503361.0, "reward": 0.7700097560882568, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8035714030265808, "reward_count_adherence_std": 0.03306501731276512, "reward_meter_mean": 0.9988847970962524, "reward_meter_std": 0.0004295332182664424, "reward_repeat_penalty_mean": 0.9593685269355774, "reward_repeat_penalty_std": 0.030346577987074852, "reward_std": 0.03843218460679054, "reward_total_composite_mean": 0.7700097560882568, "reward_total_composite_std": 0.03843219205737114, "reward_total_mean": 0.7700097560882568, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8035714030265808, "rewards/count_adherence/std": 0.03306501731276512, "rewards/meter/mean": 0.9988847970962524, "rewards/meter/std": 0.0004295332182664424, "rewards/repeat_penalty/mean": 0.9593685269355774, "rewards/repeat_penalty/std": 0.030346577987074852, "rewards/total_composite/mean": 0.7700097560882568, "rewards/total_composite/std": 0.03843219205737114, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010416030883789, "sampling/importance_sampling_ratio/min": 0.21670202910900116, "sampling/sampling_logp_difference/max": 1.5292320251464844, "sampling/sampling_logp_difference/mean": 0.05211654305458069, "step": 2865 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.0056428477400913835, "epoch": 0.1151142707956782, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.3181818181818182e-06, "loss": 0.0, "num_tokens": 6504873.0, "reward": 0.9980231523513794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9980231523513794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.015303611755371, "sampling/importance_sampling_ratio/mean": 1.0006215572357178, "sampling/importance_sampling_ratio/min": 0.9985290169715881, "sampling/sampling_logp_difference/max": 0.01518770307302475, "sampling/sampling_logp_difference/mean": 0.0006467251223511994, "step": 2866 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 65.0, "completions/max_terminated_length": 65.0, "completions/mean_length": 64.125, "completions/mean_terminated_length": 64.125, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0063372578588314354, "epoch": 0.11515443627746315, "frac_reward_zero_std": 0.0, "grad_norm": 4.305418968200684, "learning_rate": 1.315151515151515e-06, "loss": 0.005, "num_tokens": 6506642.0, "reward": 0.9991264343261719, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991264343261719, "reward_meter_std": 0.0007721302099525928, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0007721302681602538, "reward_total_composite_mean": 0.9991264343261719, "reward_total_composite_std": 0.0007721302099525928, "reward_total_mean": 0.9991264343261719, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991264343261719, "rewards/meter/std": 0.0007721302099525928, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991264343261719, "rewards/total_composite/std": 0.0007721302099525928, "sampling/importance_sampling_ratio/max": 1.0192989110946655, "sampling/importance_sampling_ratio/mean": 0.9988252520561218, "sampling/importance_sampling_ratio/min": 0.33203551173210144, "sampling/sampling_logp_difference/max": 1.102513313293457, "sampling/sampling_logp_difference/mean": 0.00290035386569798, "step": 2867 }, { "clip_ratio/high_max": 0.006192129687406123, "clip_ratio/high_mean": 0.006192129687406123, "clip_ratio/low_mean": 0.003246753243729472, "clip_ratio/low_min": 0.003246753243729472, "clip_ratio/region_mean": 0.009438882931135595, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 80.25, "completions/mean_terminated_length": 80.25, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.21316692978143692, "epoch": 0.1151946017592481, "frac_reward_zero_std": 0.0, "grad_norm": 2.3818113803863525, "learning_rate": 1.3121212121212123e-06, "loss": -0.0079, "num_tokens": 6508612.0, "reward": 0.9989343285560608, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989343285560608, "reward_meter_std": 0.00022084804368205369, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00022085075033828616, "reward_total_composite_mean": 0.9989343285560608, "reward_total_composite_std": 0.00022084804368205369, "reward_total_mean": 0.9989343285560608, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989343285560608, "rewards/meter/std": 0.00022084804368205369, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989343285560608, "rewards/total_composite/std": 0.00022084804368205369, "sampling/importance_sampling_ratio/max": 1.6601855754852295, "sampling/importance_sampling_ratio/mean": 1.0084359645843506, "sampling/importance_sampling_ratio/min": 0.3050566613674164, "sampling/sampling_logp_difference/max": 1.1872577667236328, "sampling/sampling_logp_difference/mean": 0.023430120199918747, "step": 2868 }, { "clip_ratio/high_max": 0.006416999618522823, "clip_ratio/high_mean": 0.006416999618522823, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006416999618522823, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 96.625, "completions/mean_terminated_length": 96.625, "completions/min_length": 94.0, "completions/min_terminated_length": 94.0, "entropy": 0.14486887026578188, "epoch": 0.11523476724103306, "frac_reward_zero_std": 0.0, "grad_norm": 2.163902997970581, "learning_rate": 1.3090909090909093e-06, "loss": -0.0098, "num_tokens": 6510609.0, "reward": 0.9977865219116211, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977865219116211, "reward_meter_std": 0.0006117028533481061, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006116984295658767, "reward_total_composite_mean": 0.9977865219116211, "reward_total_composite_std": 0.0006117028533481061, "reward_total_mean": 0.9977865219116211, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977865219116211, "rewards/meter/std": 0.0006117028533481061, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977865219116211, "rewards/total_composite/std": 0.0006117028533481061, "sampling/importance_sampling_ratio/max": 1.840904712677002, "sampling/importance_sampling_ratio/mean": 1.0011489391326904, "sampling/importance_sampling_ratio/min": 0.3164120018482208, "sampling/sampling_logp_difference/max": 1.150710105895996, "sampling/sampling_logp_difference/mean": 0.016453739255666733, "step": 2869 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0020491802133619785, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.05785668967291713, "epoch": 0.11527493272281801, "frac_reward_zero_std": 0.0, "grad_norm": 1.1164172887802124, "learning_rate": 1.3060606060606062e-06, "loss": -0.0, "num_tokens": 6512337.0, "reward": 0.9973324537277222, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973324537277222, "reward_meter_std": 2.97028473141836e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.9693699616473168e-05, "reward_total_composite_mean": 0.9973324537277222, "reward_total_composite_std": 2.97028473141836e-05, "reward_total_mean": 0.9973324537277222, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973324537277222, "rewards/meter/std": 2.97028473141836e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973324537277222, "rewards/total_composite/std": 2.97028473141836e-05, "sampling/importance_sampling_ratio/max": 1.6584101915359497, "sampling/importance_sampling_ratio/mean": 1.003736138343811, "sampling/importance_sampling_ratio/min": 0.6083394289016724, "sampling/sampling_logp_difference/max": 0.505859375, "sampling/sampling_logp_difference/mean": 0.010174022056162357, "step": 2870 }, { "clip_ratio/high_max": 0.01095727866049856, "clip_ratio/high_mean": 0.01095727866049856, "clip_ratio/low_mean": 0.012681030901148915, "clip_ratio/low_min": 0.012681030901148915, "clip_ratio/region_mean": 0.023638309561647475, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 79.75, "completions/mean_terminated_length": 79.75, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.22218100167810917, "epoch": 0.11531509820460296, "frac_reward_zero_std": 0.0, "grad_norm": 1.2812321186065674, "learning_rate": 1.3030303030303032e-06, "loss": 0.0021, "num_tokens": 6514415.0, "reward": 0.9988850355148315, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988850355148315, "reward_meter_std": 0.00015237655316013843, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00015238845662679523, "reward_total_composite_mean": 0.9988850355148315, "reward_total_composite_std": 0.00015237655316013843, "reward_total_mean": 0.9988850355148315, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988850355148315, "rewards/meter/std": 0.00015237655316013843, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988850355148315, "rewards/total_composite/std": 0.00015237655316013843, "sampling/importance_sampling_ratio/max": 1.7389991283416748, "sampling/importance_sampling_ratio/mean": 1.0047295093536377, "sampling/importance_sampling_ratio/min": 0.4057614207267761, "sampling/sampling_logp_difference/max": 0.9019899368286133, "sampling/sampling_logp_difference/mean": 0.02608993463218212, "step": 2871 }, { "clip_ratio/high_max": 0.0125664914958179, "clip_ratio/high_mean": 0.0125664914958179, "clip_ratio/low_mean": 0.003998056286945939, "clip_ratio/low_min": 0.003998056286945939, "clip_ratio/region_mean": 0.01656454778276384, "completions/clipped_ratio": 0.0, "completions/max_length": 251.0, "completions/max_terminated_length": 251.0, "completions/mean_length": 249.375, "completions/mean_terminated_length": 249.375, "completions/min_length": 248.0, "completions/min_terminated_length": 248.0, "entropy": 0.27019675448536873, "epoch": 0.11535526368638792, "frac_reward_zero_std": 0.0, "grad_norm": 1.5020183324813843, "learning_rate": 1.3e-06, "loss": 0.0065, "num_tokens": 6517818.0, "reward": 0.9505188465118408, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9985396862030029, "reward_meter_std": 0.0008591708028689027, "reward_repeat_penalty_mean": 0.9519230723381042, "reward_repeat_penalty_std": 0.05723259598016739, "reward_std": 0.05688953027129173, "reward_total_composite_mean": 0.9505188465118408, "reward_total_composite_std": 0.05688954144716263, "reward_total_mean": 0.9505188465118408, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9985396862030029, "rewards/meter/std": 0.0008591708028689027, "rewards/repeat_penalty/mean": 0.9519230723381042, "rewards/repeat_penalty/std": 0.05723259598016739, "rewards/total_composite/mean": 0.9505188465118408, "rewards/total_composite/std": 0.05688954144716263, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0057145357131958, "sampling/importance_sampling_ratio/min": 0.20937703549861908, "sampling/sampling_logp_difference/max": 1.5636186599731445, "sampling/sampling_logp_difference/mean": 0.025731699541211128, "step": 2872 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.0018656715983524919, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.057810590136796236, "epoch": 0.11539542916817287, "frac_reward_zero_std": 0.0, "grad_norm": 0.2703946828842163, "learning_rate": 1.296969696969697e-06, "loss": -0.0005, "num_tokens": 6519782.0, "reward": 0.9981412291526794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981412291526794, "reward_meter_std": 1.5779543900862336e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.5772715414641425e-05, "reward_total_composite_mean": 0.9981412291526794, "reward_total_composite_std": 1.5779543900862336e-05, "reward_total_mean": 0.9981412291526794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981412291526794, "rewards/meter/std": 1.5779543900862336e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981412291526794, "rewards/total_composite/std": 1.5779543900862336e-05, "sampling/importance_sampling_ratio/max": 1.4784173965454102, "sampling/importance_sampling_ratio/mean": 1.0045442581176758, "sampling/importance_sampling_ratio/min": 0.5828580856323242, "sampling/sampling_logp_difference/max": 0.5398116111755371, "sampling/sampling_logp_difference/mean": 0.009107197634875774, "step": 2873 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 37.0, "completions/max_terminated_length": 37.0, "completions/mean_length": 36.125, "completions/mean_terminated_length": 36.125, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.02826439938507974, "epoch": 0.11543559464995783, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.2939393939393941e-06, "loss": 0.0, "num_tokens": 6521271.0, "reward": 0.9996045231819153, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9996045231819153, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0343594551086426, "sampling/importance_sampling_ratio/mean": 0.9987858533859253, "sampling/importance_sampling_ratio/min": 0.6088523864746094, "sampling/sampling_logp_difference/max": 0.496179461479187, "sampling/sampling_logp_difference/mean": 0.0043750000186264515, "step": 2874 }, { "clip_ratio/high_max": 0.007520091836340725, "clip_ratio/high_mean": 0.007520091836340725, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/region_mean": 0.009331686072982848, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.17508152220398188, "epoch": 0.11547576013174278, "frac_reward_zero_std": 0.0, "grad_norm": 5.8196916580200195, "learning_rate": 1.290909090909091e-06, "loss": 0.0072, "num_tokens": 6523051.0, "reward": 0.9894192218780518, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9894192218780518, "reward_meter_std": 0.010708401910960674, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01070839911699295, "reward_total_composite_mean": 0.9894192218780518, "reward_total_composite_std": 0.010708401910960674, "reward_total_mean": 0.9894192218780518, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9894192218780518, "rewards/meter/std": 0.010708401910960674, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9894192218780518, "rewards/total_composite/std": 0.010708401910960674, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0059653520584106, "sampling/importance_sampling_ratio/min": 0.10831031203269958, "sampling/sampling_logp_difference/max": 2.222754955291748, "sampling/sampling_logp_difference/mean": 0.02634219266474247, "step": 2875 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/region_mean": 0.006147540640085936, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.05711106630042195, "epoch": 0.11551592561352773, "frac_reward_zero_std": 0.0, "grad_norm": 0.7274681925773621, "learning_rate": 1.287878787878788e-06, "loss": -0.0001, "num_tokens": 6525011.0, "reward": 0.9973265528678894, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973265528678894, "reward_meter_std": 2.102993312291801e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.101841710100416e-05, "reward_total_composite_mean": 0.9973265528678894, "reward_total_composite_std": 2.102993312291801e-05, "reward_total_mean": 0.9973265528678894, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973265528678894, "rewards/meter/std": 2.102993312291801e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973265528678894, "rewards/total_composite/std": 2.102993312291801e-05, "sampling/importance_sampling_ratio/max": 1.5096032619476318, "sampling/importance_sampling_ratio/mean": 0.9998522400856018, "sampling/importance_sampling_ratio/min": 0.41592586040496826, "sampling/sampling_logp_difference/max": 0.8772482872009277, "sampling/sampling_logp_difference/mean": 0.010243063792586327, "step": 2876 }, { "clip_ratio/high_max": 0.023920452571474016, "clip_ratio/high_mean": 0.023920452571474016, "clip_ratio/low_mean": 0.015123688150197268, "clip_ratio/low_min": 0.015123688150197268, "clip_ratio/region_mean": 0.03904414072167128, "completions/clipped_ratio": 0.0, "completions/max_length": 123.0, "completions/max_terminated_length": 123.0, "completions/mean_length": 118.375, "completions/mean_terminated_length": 118.375, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.35712629184126854, "epoch": 0.11555609109531269, "frac_reward_zero_std": 0.0, "grad_norm": 2.641002655029297, "learning_rate": 1.2848484848484848e-06, "loss": -0.0056, "num_tokens": 6527326.0, "reward": 0.9990575313568115, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990575313568115, "reward_meter_std": 0.00029957492370158434, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00029957492370158434, "reward_total_composite_mean": 0.9990575313568115, "reward_total_composite_std": 0.00029957492370158434, "reward_total_mean": 0.9990575313568115, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990575313568115, "rewards/meter/std": 0.00029957492370158434, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990575313568115, "rewards/total_composite/std": 0.00029957492370158434, "sampling/importance_sampling_ratio/max": 1.9006513357162476, "sampling/importance_sampling_ratio/mean": 1.0039846897125244, "sampling/importance_sampling_ratio/min": 0.14651189744472504, "sampling/sampling_logp_difference/max": 1.9206485748291016, "sampling/sampling_logp_difference/mean": 0.0404638797044754, "step": 2877 }, { "clip_ratio/high_max": 0.030017988989129663, "clip_ratio/high_mean": 0.030017988989129663, "clip_ratio/low_mean": 0.012780054472386837, "clip_ratio/low_min": 0.012780054472386837, "clip_ratio/region_mean": 0.0427980434615165, "completions/clipped_ratio": 0.0, "completions/max_length": 310.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 302.375, "completions/mean_terminated_length": 302.375, "completions/min_length": 294.0, "completions/min_terminated_length": 294.0, "entropy": 0.47275901213288307, "epoch": 0.11559625657709764, "frac_reward_zero_std": 0.0, "grad_norm": 2.2012033462524414, "learning_rate": 1.2818181818181819e-06, "loss": 0.0099, "num_tokens": 6531545.0, "reward": 0.975556492805481, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9952223896980286, "reward_meter_std": 0.0032623247243463993, "reward_repeat_penalty_mean": 0.9802631139755249, "reward_repeat_penalty_std": 0.027239417657256126, "reward_std": 0.026347849518060684, "reward_total_composite_mean": 0.975556492805481, "reward_total_composite_std": 0.02634783275425434, "reward_total_mean": 0.975556492805481, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9952223896980286, "rewards/meter/std": 0.0032623247243463993, "rewards/repeat_penalty/mean": 0.9802631139755249, "rewards/repeat_penalty/std": 0.027239417657256126, "rewards/total_composite/mean": 0.975556492805481, "rewards/total_composite/std": 0.02634783275425434, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0102460384368896, "sampling/importance_sampling_ratio/min": 0.23934902250766754, "sampling/sampling_logp_difference/max": 1.4298324584960938, "sampling/sampling_logp_difference/mean": 0.05190505087375641, "step": 2878 }, { "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/low_mean": 0.003839680110104382, "clip_ratio/low_min": 0.003839680110104382, "clip_ratio/region_mean": 0.006390700465999544, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.75, "completions/mean_terminated_length": 97.75, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.04796775570139289, "epoch": 0.1156364220588826, "frac_reward_zero_std": 0.0, "grad_norm": 0.22587938606739044, "learning_rate": 1.2787878787878787e-06, "loss": -0.0009, "num_tokens": 6533695.0, "reward": 0.9994171857833862, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994171857833862, "reward_meter_std": 3.359419133630581e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.360588743817061e-05, "reward_total_composite_mean": 0.9994171857833862, "reward_total_composite_std": 3.359419133630581e-05, "reward_total_mean": 0.9994171857833862, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994171857833862, "rewards/meter/std": 3.359419133630581e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994171857833862, "rewards/total_composite/std": 3.359419133630581e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0023701190948486, "sampling/importance_sampling_ratio/min": 0.569092333316803, "sampling/sampling_logp_difference/max": 2.2170257568359375, "sampling/sampling_logp_difference/mean": 0.009699815884232521, "step": 2879 }, { "clip_ratio/high_max": 0.01734496164135635, "clip_ratio/high_mean": 0.01734496164135635, "clip_ratio/low_mean": 0.014072955353185534, "clip_ratio/low_min": 0.014072955353185534, "clip_ratio/region_mean": 0.03141791699454188, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 131.25, "completions/mean_terminated_length": 131.25, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.46811263263225555, "epoch": 0.11567658754066755, "frac_reward_zero_std": 0.0, "grad_norm": 3.3986284732818604, "learning_rate": 1.2757575757575758e-06, "loss": 0.0276, "num_tokens": 6536129.0, "reward": 0.9821168184280396, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9821168184280396, "reward_meter_std": 0.010360779240727425, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010360777378082275, "reward_total_composite_mean": 0.9821168184280396, "reward_total_composite_std": 0.010360779240727425, "reward_total_mean": 0.9821168184280396, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9821168184280396, "rewards/meter/std": 0.010360779240727425, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9821168184280396, "rewards/total_composite/std": 0.010360779240727425, "sampling/importance_sampling_ratio/max": 1.8226743936538696, "sampling/importance_sampling_ratio/mean": 1.0092159509658813, "sampling/importance_sampling_ratio/min": 0.3004924952983856, "sampling/sampling_logp_difference/max": 1.2023324966430664, "sampling/sampling_logp_difference/mean": 0.045867305248975754, "step": 2880 }, { "clip_ratio/high_max": 0.0038528296863660216, "clip_ratio/high_mean": 0.0038528296863660216, "clip_ratio/low_mean": 0.005141489440575242, "clip_ratio/low_min": 0.005141489440575242, "clip_ratio/region_mean": 0.008994319126941264, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.375, "completions/mean_terminated_length": 97.375, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.1293118530884385, "epoch": 0.1157167530224525, "frac_reward_zero_std": 0.0, "grad_norm": 0.9868201613426208, "learning_rate": 1.2727272727272728e-06, "loss": -0.0002, "num_tokens": 6538228.0, "reward": 0.9979463815689087, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979463815689087, "reward_meter_std": 6.646273686783388e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.644289533142e-05, "reward_total_composite_mean": 0.9979463815689087, "reward_total_composite_std": 6.646273686783388e-05, "reward_total_mean": 0.9979463815689087, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979463815689087, "rewards/meter/std": 6.646273686783388e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979463815689087, "rewards/total_composite/std": 6.646273686783388e-05, "sampling/importance_sampling_ratio/max": 1.4697163105010986, "sampling/importance_sampling_ratio/mean": 1.0037821531295776, "sampling/importance_sampling_ratio/min": 0.4541665017604828, "sampling/sampling_logp_difference/max": 0.7892913818359375, "sampling/sampling_logp_difference/mean": 0.013669819571077824, "step": 2881 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0014212291716830805, "epoch": 0.11575691850423746, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.2696969696969698e-06, "loss": 0.0, "num_tokens": 6540020.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0051120519638062, "sampling/importance_sampling_ratio/mean": 1.0001249313354492, "sampling/importance_sampling_ratio/min": 0.9974856972694397, "sampling/sampling_logp_difference/max": 0.005098958499729633, "sampling/sampling_logp_difference/mean": 0.00015525527123827487, "step": 2882 }, { "clip_ratio/high_max": 0.013993188505992293, "clip_ratio/high_mean": 0.013993188505992293, "clip_ratio/low_mean": 0.004407651489600539, "clip_ratio/low_min": 0.004407651489600539, "clip_ratio/region_mean": 0.018400839995592833, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 142.5, "completions/mean_terminated_length": 142.5, "completions/min_length": 141.0, "completions/min_terminated_length": 141.0, "entropy": 0.2192006167024374, "epoch": 0.11579708398602241, "frac_reward_zero_std": 0.0, "grad_norm": 1.7158304452896118, "learning_rate": 1.2666666666666669e-06, "loss": -0.0014, "num_tokens": 6542928.0, "reward": 0.998982846736908, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998982846736908, "reward_meter_std": 0.00020013147150166333, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020013845642097294, "reward_total_composite_mean": 0.998982846736908, "reward_total_composite_std": 0.00020013147150166333, "reward_total_mean": 0.998982846736908, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998982846736908, "rewards/meter/std": 0.00020013147150166333, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998982846736908, "rewards/total_composite/std": 0.00020013147150166333, "sampling/importance_sampling_ratio/max": 1.6984710693359375, "sampling/importance_sampling_ratio/mean": 1.0040578842163086, "sampling/importance_sampling_ratio/min": 0.3756113350391388, "sampling/sampling_logp_difference/max": 0.9792003631591797, "sampling/sampling_logp_difference/mean": 0.022738341242074966, "step": 2883 }, { "clip_ratio/high_max": 0.014788191299885511, "clip_ratio/high_mean": 0.014788191299885511, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.014788191299885511, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.625, "completions/mean_terminated_length": 67.625, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.14564206078648567, "epoch": 0.11583724946780737, "frac_reward_zero_std": 0.0, "grad_norm": 2.683151960372925, "learning_rate": 1.2636363636363637e-06, "loss": 0.0006, "num_tokens": 6544901.0, "reward": 0.9992280006408691, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992280006408691, "reward_meter_std": 0.00013651978224515915, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00013652884808834642, "reward_total_composite_mean": 0.9992280006408691, "reward_total_composite_std": 0.00013651978224515915, "reward_total_mean": 0.9992280006408691, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992280006408691, "rewards/meter/std": 0.00013651978224515915, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992280006408691, "rewards/total_composite/std": 0.00013651978224515915, "sampling/importance_sampling_ratio/max": 1.5717170238494873, "sampling/importance_sampling_ratio/mean": 0.999349057674408, "sampling/importance_sampling_ratio/min": 0.3337857127189636, "sampling/sampling_logp_difference/max": 1.097256064414978, "sampling/sampling_logp_difference/mean": 0.019809167832136154, "step": 2884 }, { "clip_ratio/high_max": 0.0034966744715347886, "clip_ratio/high_mean": 0.0034966744715347886, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/region_mean": 0.0052327855955809355, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.08360666781663895, "epoch": 0.11587741494959232, "frac_reward_zero_std": 0.0, "grad_norm": 0.26706913113594055, "learning_rate": 1.2606060606060608e-06, "loss": 0.0008, "num_tokens": 6546912.0, "reward": 0.9994363784790039, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994363784790039, "reward_meter_std": 2.7232801585341804e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.7221845812164247e-05, "reward_total_composite_mean": 0.9994363784790039, "reward_total_composite_std": 2.7232801585341804e-05, "reward_total_mean": 0.9994363784790039, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994363784790039, "rewards/meter/std": 2.7232801585341804e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994363784790039, "rewards/total_composite/std": 2.7232801585341804e-05, "sampling/importance_sampling_ratio/max": 1.2115890979766846, "sampling/importance_sampling_ratio/mean": 1.002943754196167, "sampling/importance_sampling_ratio/min": 0.5084663033485413, "sampling/sampling_logp_difference/max": 0.676356315612793, "sampling/sampling_logp_difference/mean": 0.010549667291343212, "step": 2885 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.005542142200283706, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.09443793492391706, "epoch": 0.11591758043137727, "frac_reward_zero_std": 0.0, "grad_norm": 2.8855912685394287, "learning_rate": 1.2575757575757578e-06, "loss": 0.0053, "num_tokens": 6548792.0, "reward": 0.9954104423522949, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9954104423522949, "reward_meter_std": 0.007760278414934874, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007760272361338139, "reward_total_composite_mean": 0.9954104423522949, "reward_total_composite_std": 0.007760278414934874, "reward_total_mean": 0.9954104423522949, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9954104423522949, "rewards/meter/std": 0.007760278414934874, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9954104423522949, "rewards/total_composite/std": 0.007760278414934874, "sampling/importance_sampling_ratio/max": 1.3913602828979492, "sampling/importance_sampling_ratio/mean": 1.0019506216049194, "sampling/importance_sampling_ratio/min": 0.39016932249069214, "sampling/sampling_logp_difference/max": 0.9411745071411133, "sampling/sampling_logp_difference/mean": 0.009755668230354786, "step": 2886 }, { "clip_ratio/high_max": 0.007520923274569213, "clip_ratio/high_mean": 0.007520923274569213, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.0094148627249524, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 65.875, "completions/mean_terminated_length": 65.875, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.17512462474405766, "epoch": 0.11595774591316223, "frac_reward_zero_std": 0.0, "grad_norm": 3.4663033485412598, "learning_rate": 1.2545454545454546e-06, "loss": -0.0051, "num_tokens": 6550591.0, "reward": 0.9924707412719727, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9924707412719727, "reward_meter_std": 0.0012687068665400147, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001268712687306106, "reward_total_composite_mean": 0.9924707412719727, "reward_total_composite_std": 0.0012687068665400147, "reward_total_mean": 0.9924707412719727, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9924707412719727, "rewards/meter/std": 0.0012687068665400147, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924707412719727, "rewards/total_composite/std": 0.0012687068665400147, "sampling/importance_sampling_ratio/max": 1.3884730339050293, "sampling/importance_sampling_ratio/mean": 1.0042961835861206, "sampling/importance_sampling_ratio/min": 0.4974890947341919, "sampling/sampling_logp_difference/max": 0.6981816291809082, "sampling/sampling_logp_difference/mean": 0.02419329807162285, "step": 2887 }, { "clip_ratio/high_max": 0.011493272730149329, "clip_ratio/high_mean": 0.011493272730149329, "clip_ratio/low_mean": 0.003839680110104382, "clip_ratio/low_min": 0.003839680110104382, "clip_ratio/region_mean": 0.01533295284025371, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 97.5, "completions/mean_terminated_length": 97.5, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.1272029634565115, "epoch": 0.11599791139494718, "frac_reward_zero_std": 0.0, "grad_norm": 0.5309938192367554, "learning_rate": 1.2515151515151517e-06, "loss": -0.001, "num_tokens": 6552707.0, "reward": 0.9980380535125732, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980380535125732, "reward_meter_std": 4.4257390982238576e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.425693259690888e-05, "reward_total_composite_mean": 0.9980380535125732, "reward_total_composite_std": 4.4257390982238576e-05, "reward_total_mean": 0.9980380535125732, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980380535125732, "rewards/meter/std": 4.4257390982238576e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980380535125732, "rewards/total_composite/std": 4.4257390982238576e-05, "sampling/importance_sampling_ratio/max": 1.4897899627685547, "sampling/importance_sampling_ratio/mean": 1.001752257347107, "sampling/importance_sampling_ratio/min": 0.4008263647556305, "sampling/sampling_logp_difference/max": 0.9142270088195801, "sampling/sampling_logp_difference/mean": 0.015023077838122845, "step": 2888 }, { "clip_ratio/high_max": 0.007871835259720683, "clip_ratio/high_mean": 0.007871835259720683, "clip_ratio/low_mean": 0.004707766929641366, "clip_ratio/low_min": 0.004707766929641366, "clip_ratio/region_mean": 0.012579602189362049, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 79.5, "completions/mean_terminated_length": 79.5, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.21664478071033955, "epoch": 0.11603807687673214, "frac_reward_zero_std": 0.0, "grad_norm": 1.1036797761917114, "learning_rate": 1.2484848484848485e-06, "loss": -0.0002, "num_tokens": 6554583.0, "reward": 0.998902440071106, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998902440071106, "reward_meter_std": 0.00011711295519489795, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011710206308634952, "reward_total_composite_mean": 0.998902440071106, "reward_total_composite_std": 0.00011711295519489795, "reward_total_mean": 0.998902440071106, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998902440071106, "rewards/meter/std": 0.00011711295519489795, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998902440071106, "rewards/total_composite/std": 0.00011711295519489795, "sampling/importance_sampling_ratio/max": 1.4718900918960571, "sampling/importance_sampling_ratio/mean": 1.008751392364502, "sampling/importance_sampling_ratio/min": 0.17142720520496368, "sampling/sampling_logp_difference/max": 1.763596534729004, "sampling/sampling_logp_difference/mean": 0.02463226206600666, "step": 2889 }, { "clip_ratio/high_max": 0.013470079400576651, "clip_ratio/high_mean": 0.013470079400576651, "clip_ratio/low_mean": 0.005469039548188448, "clip_ratio/low_min": 0.005469039548188448, "clip_ratio/region_mean": 0.0189391189487651, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 92.0, "completions/mean_terminated_length": 92.0, "completions/min_length": 87.0, "completions/min_terminated_length": 87.0, "entropy": 0.14203907828778028, "epoch": 0.11607824235851709, "frac_reward_zero_std": 0.0, "grad_norm": 2.1610708236694336, "learning_rate": 1.2454545454545456e-06, "loss": -0.0123, "num_tokens": 6556527.0, "reward": 0.9974758625030518, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974758625030518, "reward_meter_std": 0.0004072287702001631, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004072203009855002, "reward_total_composite_mean": 0.9974758625030518, "reward_total_composite_std": 0.0004072287702001631, "reward_total_mean": 0.9974758625030518, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974758625030518, "rewards/meter/std": 0.0004072287702001631, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9974758625030518, "rewards/total_composite/std": 0.0004072287702001631, "sampling/importance_sampling_ratio/max": 1.664230227470398, "sampling/importance_sampling_ratio/mean": 1.0036619901657104, "sampling/importance_sampling_ratio/min": 0.26348739862442017, "sampling/sampling_logp_difference/max": 1.333749771118164, "sampling/sampling_logp_difference/mean": 0.01821918971836567, "step": 2890 }, { "clip_ratio/high_max": 0.01412348123267293, "clip_ratio/high_mean": 0.01412348123267293, "clip_ratio/low_mean": 0.008739139186218381, "clip_ratio/low_min": 0.008739139186218381, "clip_ratio/region_mean": 0.02286262041889131, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 97.75, "completions/mean_terminated_length": 97.75, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.2828730084002018, "epoch": 0.11611840784030204, "frac_reward_zero_std": 0.0, "grad_norm": 3.022761106491089, "learning_rate": 1.2424242424242424e-06, "loss": 0.0102, "num_tokens": 6558597.0, "reward": 0.9901492595672607, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9901492595672607, "reward_meter_std": 0.005893507041037083, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005893507041037083, "reward_total_composite_mean": 0.9901492595672607, "reward_total_composite_std": 0.005893507041037083, "reward_total_mean": 0.9901492595672607, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9901492595672607, "rewards/meter/std": 0.005893507041037083, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9901492595672607, "rewards/total_composite/std": 0.005893507041037083, "sampling/importance_sampling_ratio/max": 1.6783695220947266, "sampling/importance_sampling_ratio/mean": 1.0117164850234985, "sampling/importance_sampling_ratio/min": 0.2823234498500824, "sampling/sampling_logp_difference/max": 1.2647018432617188, "sampling/sampling_logp_difference/mean": 0.032126907259225845, "step": 2891 }, { "clip_ratio/high_max": 0.06117365509271622, "clip_ratio/high_mean": 0.06117365509271622, "clip_ratio/low_mean": 0.020869565196335316, "clip_ratio/low_min": 0.020869565196335316, "clip_ratio/region_mean": 0.08204322028905153, "completions/clipped_ratio": 0.0, "completions/max_length": 25.0, "completions/max_terminated_length": 25.0, "completions/mean_length": 23.5, "completions/mean_terminated_length": 23.5, "completions/min_length": 21.0, "completions/min_terminated_length": 21.0, "entropy": 0.37175317015498877, "epoch": 0.116158573322087, "frac_reward_zero_std": 0.0, "grad_norm": 15.570990562438965, "learning_rate": 1.2393939393939394e-06, "loss": 0.0466, "num_tokens": 6559937.0, "reward": 0.895869255065918, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.895869255065918, "reward_meter_std": 0.033730946481227875, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03373093158006668, "reward_total_composite_mean": 0.895869255065918, "reward_total_composite_std": 0.033730946481227875, "reward_total_mean": 0.895869255065918, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.895869255065918, "rewards/meter/std": 0.033730946481227875, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.895869255065918, "rewards/total_composite/std": 0.033730946481227875, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9834393858909607, "sampling/importance_sampling_ratio/min": 0.24084904789924622, "sampling/sampling_logp_difference/max": 1.4235849380493164, "sampling/sampling_logp_difference/mean": 0.0897468775510788, "step": 2892 }, { "clip_ratio/high_max": 0.010245901066809893, "clip_ratio/high_mean": 0.010245901066809893, "clip_ratio/low_mean": 0.006696428870782256, "clip_ratio/low_min": 0.006696428870782256, "clip_ratio/region_mean": 0.01694232993759215, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 59.75, "completions/mean_terminated_length": 59.75, "completions/min_length": 56.0, "completions/min_terminated_length": 56.0, "entropy": 0.07222167262807488, "epoch": 0.11619873880387195, "frac_reward_zero_std": 0.0, "grad_norm": 11.483199119567871, "learning_rate": 1.2363636363636365e-06, "loss": -0.0319, "num_tokens": 6561687.0, "reward": 0.9967870712280273, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9967870712280273, "reward_meter_std": 0.0010188753949478269, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0010188753949478269, "reward_total_composite_mean": 0.9967870712280273, "reward_total_composite_std": 0.0010188753949478269, "reward_total_mean": 0.9967870712280273, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9967870712280273, "rewards/meter/std": 0.0010188753949478269, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9967870712280273, "rewards/total_composite/std": 0.0010188753949478269, "sampling/importance_sampling_ratio/max": 1.8504060506820679, "sampling/importance_sampling_ratio/mean": 1.0000895261764526, "sampling/importance_sampling_ratio/min": 0.17961078882217407, "sampling/sampling_logp_difference/max": 1.7169630527496338, "sampling/sampling_logp_difference/mean": 0.01920030452311039, "step": 2893 }, { "clip_ratio/high_max": 0.025422006146982312, "clip_ratio/high_mean": 0.025422006146982312, "clip_ratio/low_mean": 0.007427186123095453, "clip_ratio/low_min": 0.007427186123095453, "clip_ratio/region_mean": 0.032849192270077765, "completions/clipped_ratio": 0.0, "completions/max_length": 166.0, "completions/max_terminated_length": 166.0, "completions/mean_length": 154.5, "completions/mean_terminated_length": 154.5, "completions/min_length": 108.0, "completions/min_terminated_length": 108.0, "entropy": 0.5260906405746937, "epoch": 0.1162389042856569, "frac_reward_zero_std": 0.0, "grad_norm": 1.820000171661377, "learning_rate": 1.2333333333333335e-06, "loss": -0.0994, "num_tokens": 6564523.0, "reward": 0.9117738604545593, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.1414213478565216, "reward_meter_mean": 0.9889054298400879, "reward_meter_std": 0.003125808434560895, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.13682788610458374, "reward_total_composite_mean": 0.9117738604545593, "reward_total_composite_std": 0.13682788610458374, "reward_total_mean": 0.9117738604545593, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.1414213478565216, "rewards/meter/mean": 0.9889054298400879, "rewards/meter/std": 0.003125808434560895, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9117738604545593, "rewards/total_composite/std": 0.13682788610458374, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.008368730545044, "sampling/importance_sampling_ratio/min": 0.24550971388816833, "sampling/sampling_logp_difference/max": 1.404418706893921, "sampling/sampling_logp_difference/mean": 0.04758564755320549, "step": 2894 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 132.0, "completions/mean_terminated_length": 132.0, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.005916567693930119, "epoch": 0.11627906976744186, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.2303030303030304e-06, "loss": 0.0, "num_tokens": 6567027.0, "reward": 0.5199131369590759, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6684597730636597, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 0.7777777910232544, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.5199131369590759, "reward_total_composite_std": 0.0, "reward_total_mean": 0.5199131369590759, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6684597730636597, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 0.7777777910232544, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.5199131369590759, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0248000621795654, "sampling/importance_sampling_ratio/mean": 1.0006672143936157, "sampling/importance_sampling_ratio/min": 0.9979704022407532, "sampling/sampling_logp_difference/max": 0.02449747547507286, "sampling/sampling_logp_difference/mean": 0.0006711529567837715, "step": 2895 }, { "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.004098360426723957, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.035036823246628046, "epoch": 0.11631923524922681, "frac_reward_zero_std": 0.0, "grad_norm": 3.799595355987549, "learning_rate": 1.2272727272727274e-06, "loss": 0.0006, "num_tokens": 6568899.0, "reward": 0.9972846508026123, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972846508026123, "reward_meter_std": 0.00015232501027639955, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00015233403246384114, "reward_total_composite_mean": 0.9972846508026123, "reward_total_composite_std": 0.00015232501027639955, "reward_total_mean": 0.9972846508026123, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972846508026123, "rewards/meter/std": 0.00015232501027639955, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972846508026123, "rewards/total_composite/std": 0.00015232501027639955, "sampling/importance_sampling_ratio/max": 1.18209969997406, "sampling/importance_sampling_ratio/mean": 1.0012083053588867, "sampling/importance_sampling_ratio/min": 0.5785737633705139, "sampling/sampling_logp_difference/max": 0.5471892356872559, "sampling/sampling_logp_difference/mean": 0.005107899196445942, "step": 2896 }, { "clip_ratio/high_max": 0.008736189920455217, "clip_ratio/high_mean": 0.008736189920455217, "clip_ratio/low_mean": 0.005829093977808952, "clip_ratio/low_min": 0.005829093977808952, "clip_ratio/region_mean": 0.01456528389826417, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 128.5, "completions/mean_terminated_length": 128.5, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.17375356331467628, "epoch": 0.11635940073101177, "frac_reward_zero_std": 0.0, "grad_norm": 1.706589937210083, "learning_rate": 1.2242424242424242e-06, "loss": 0.0028, "num_tokens": 6571375.0, "reward": 0.9259439706802368, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972264766693115, "reward_meter_std": 0.002113205147907138, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.07545064389705658, "reward_total_composite_mean": 0.9259439706802368, "reward_total_composite_std": 0.07545064389705658, "reward_total_mean": 0.9259439706802368, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972264766693115, "rewards/meter/std": 0.002113205147907138, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9259439706802368, "rewards/total_composite/std": 0.07545064389705658, "sampling/importance_sampling_ratio/max": 1.4906891584396362, "sampling/importance_sampling_ratio/mean": 1.0051052570343018, "sampling/importance_sampling_ratio/min": 0.4062493145465851, "sampling/sampling_logp_difference/max": 0.9007883071899414, "sampling/sampling_logp_difference/mean": 0.016268158331513405, "step": 2897 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0020958341483492404, "epoch": 0.11639956621279672, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.2212121212121213e-06, "loss": 0.0, "num_tokens": 6573071.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0040686130523682, "sampling/importance_sampling_ratio/mean": 1.0002572536468506, "sampling/importance_sampling_ratio/min": 0.9999027252197266, "sampling/sampling_logp_difference/max": 0.004060388542711735, "sampling/sampling_logp_difference/mean": 0.0002584453613962978, "step": 2898 }, { "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/region_mean": 0.008196720853447914, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.034302944084629416, "epoch": 0.11643973169458167, "frac_reward_zero_std": 0.0, "grad_norm": 0.039646849036216736, "learning_rate": 1.2181818181818183e-06, "loss": -0.0001, "num_tokens": 6574871.0, "reward": 0.9973360300064087, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973360300064087, "reward_meter_std": 8.113272997434251e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.10424353403505e-06, "reward_total_composite_mean": 0.9973360300064087, "reward_total_composite_std": 8.113272997434251e-06, "reward_total_mean": 0.9973360300064087, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973360300064087, "rewards/meter/std": 8.113272997434251e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973360300064087, "rewards/total_composite/std": 8.113272997434251e-06, "sampling/importance_sampling_ratio/max": 1.1416552066802979, "sampling/importance_sampling_ratio/mean": 1.0008740425109863, "sampling/importance_sampling_ratio/min": 0.617810070514679, "sampling/sampling_logp_difference/max": 0.4815742075443268, "sampling/sampling_logp_difference/mean": 0.004976366646587849, "step": 2899 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0027679620689013973, "epoch": 0.11647989717636663, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.2151515151515154e-06, "loss": 0.0, "num_tokens": 6576751.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.006137490272522, "sampling/importance_sampling_ratio/mean": 1.000099539756775, "sampling/importance_sampling_ratio/min": 0.9910017251968384, "sampling/sampling_logp_difference/max": 0.009038996882736683, "sampling/sampling_logp_difference/mean": 0.0002209599915659055, "step": 2900 }, { "epoch": 0.11647989717636663, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/max_length": 422.84615384615387, "eval_completions/max_terminated_length": 402.46153846153845, "eval_completions/mean_length": 211.68269230769232, "eval_completions/mean_terminated_length": 205.16483600323016, "eval_completions/min_length": 61.84615384615385, "eval_completions/min_terminated_length": 61.84615384615385, "eval_entropy": 0.4129388790864211, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 6576751.0, "eval_reward": 0.7234287353662344, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.9613341138913081, "eval_reward_count_adherence_std": 0.06466435583738181, "eval_reward_meter_mean": 0.7883604077192453, "eval_reward_meter_std": 0.34034269847548926, "eval_reward_repeat_penalty_mean": 0.955476293197045, "eval_reward_repeat_penalty_std": 0.07249205201291122, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7234287353662344, "eval_reward_total_composite_std": 0.343703310077007, "eval_reward_total_mean": 0.7234287353662344, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.9613341138913081, "eval_rewards/count_adherence/std": 0.06466435583738181, "eval_rewards/meter/mean": 0.7883604077192453, "eval_rewards/meter/std": 0.34034269847548926, "eval_rewards/repeat_penalty/mean": 0.955476293197045, "eval_rewards/repeat_penalty/std": 0.07249205201291122, "eval_rewards/total_composite/mean": 0.7234287353662344, "eval_rewards/total_composite/std": 0.343703310077007, "eval_runtime": 78.4008, "eval_samples_per_second": 1.327, "eval_sampling/importance_sampling_ratio/max": 1.529663553604713, "eval_sampling/importance_sampling_ratio/mean": 1.0095937985640306, "eval_sampling/importance_sampling_ratio/min": 0.3125262191662422, "eval_sampling/sampling_logp_difference/max": 1.1897428219134991, "eval_sampling/sampling_logp_difference/mean": 0.03452126896725251, "eval_steps_per_second": 0.166, "step": 2900 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.005171062133740634, "epoch": 0.11652006265815158, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.2121212121212122e-06, "loss": 0.0, "num_tokens": 6578415.0, "reward": 0.9980231523513794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9980231523513794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0070579051971436, "sampling/importance_sampling_ratio/mean": 1.0005377531051636, "sampling/importance_sampling_ratio/min": 0.9922874569892883, "sampling/sampling_logp_difference/max": 0.007742481306195259, "sampling/sampling_logp_difference/mean": 0.0006013626698404551, "step": 2901 }, { "clip_ratio/high_max": 0.010358819272369146, "clip_ratio/high_mean": 0.010358819272369146, "clip_ratio/low_mean": 0.010757095878943801, "clip_ratio/low_min": 0.010757095878943801, "clip_ratio/region_mean": 0.021115915151312947, "completions/clipped_ratio": 0.25, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 501.25, "completions/mean_terminated_length": 497.66668701171875, "completions/min_length": 485.0, "completions/min_terminated_length": 485.0, "entropy": 0.3398243226110935, "epoch": 0.11656022813993654, "frac_reward_zero_std": 0.0, "grad_norm": 13.265780448913574, "learning_rate": 1.2090909090909092e-06, "loss": 0.2202, "num_tokens": 6582881.0, "reward": 0.7718735337257385, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.796875, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9986035823822021, "reward_meter_std": 0.0011026532156392932, "reward_repeat_penalty_mean": 0.9701460003852844, "reward_repeat_penalty_std": 0.01849004067480564, "reward_std": 0.042419061064720154, "reward_total_composite_mean": 0.7718735337257385, "reward_total_composite_std": 0.04241907224059105, "reward_total_mean": 0.7718735337257385, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.796875, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9986035823822021, "rewards/meter/std": 0.0011026532156392932, "rewards/repeat_penalty/mean": 0.9701460003852844, "rewards/repeat_penalty/std": 0.01849004067480564, "rewards/total_composite/mean": 0.7718735337257385, "rewards/total_composite/std": 0.04241907224059105, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0130130052566528, "sampling/importance_sampling_ratio/min": 0.07789622992277145, "sampling/sampling_logp_difference/max": 2.552377700805664, "sampling/sampling_logp_difference/mean": 0.04629309102892876, "step": 2902 }, { "clip_ratio/high_max": 0.01695435866713524, "clip_ratio/high_mean": 0.01695435866713524, "clip_ratio/low_mean": 0.025543570518493652, "clip_ratio/low_min": 0.025543570518493652, "clip_ratio/region_mean": 0.04249792918562889, "completions/clipped_ratio": 0.0, "completions/max_length": 349.0, "completions/max_terminated_length": 349.0, "completions/mean_length": 325.625, "completions/mean_terminated_length": 325.625, "completions/min_length": 304.0, "completions/min_terminated_length": 304.0, "entropy": 0.633208267390728, "epoch": 0.11660039362172149, "frac_reward_zero_std": 0.0, "grad_norm": 2.858194351196289, "learning_rate": 1.206060606060606e-06, "loss": -0.0437, "num_tokens": 6587110.0, "reward": 0.9380475282669067, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.0534522607922554, "reward_meter_mean": 0.9878900051116943, "reward_meter_std": 0.025390086695551872, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.049617357552051544, "reward_total_composite_mean": 0.9380475282669067, "reward_total_composite_std": 0.04961733520030975, "reward_total_mean": 0.9380475282669067, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.0534522607922554, "rewards/meter/mean": 0.9878900051116943, "rewards/meter/std": 0.025390086695551872, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9380475282669067, "rewards/total_composite/std": 0.04961733520030975, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.017285943031311, "sampling/importance_sampling_ratio/min": 0.0775475725531578, "sampling/sampling_logp_difference/max": 2.556863784790039, "sampling/sampling_logp_difference/mean": 0.06375819444656372, "step": 2903 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0017551528289914131, "epoch": 0.11664055910350644, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.2030303030303031e-06, "loss": 0.0, "num_tokens": 6588910.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0044045448303223, "sampling/importance_sampling_ratio/mean": 1.0000479221343994, "sampling/importance_sampling_ratio/min": 0.9847897291183472, "sampling/sampling_logp_difference/max": 0.01532711274921894, "sampling/sampling_logp_difference/mean": 0.00021447156905196607, "step": 2904 }, { "clip_ratio/high_max": 0.008968020090833306, "clip_ratio/high_mean": 0.008968020090833306, "clip_ratio/low_mean": 0.005181760177947581, "clip_ratio/low_min": 0.005181760177947581, "clip_ratio/region_mean": 0.014149780268780887, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.5, "completions/mean_terminated_length": 97.5, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.11437637638300657, "epoch": 0.1166807245852914, "frac_reward_zero_std": 0.0, "grad_norm": 0.42203962802886963, "learning_rate": 1.2000000000000002e-06, "loss": 0.0003, "num_tokens": 6591138.0, "reward": 0.9980453252792358, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980453252792358, "reward_meter_std": 4.257191903889179e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.25745893153362e-05, "reward_total_composite_mean": 0.9980453252792358, "reward_total_composite_std": 4.257191903889179e-05, "reward_total_mean": 0.9980453252792358, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980453252792358, "rewards/meter/std": 4.257191903889179e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980453252792358, "rewards/total_composite/std": 4.257191903889179e-05, "sampling/importance_sampling_ratio/max": 1.5006216764450073, "sampling/importance_sampling_ratio/mean": 1.002492904663086, "sampling/importance_sampling_ratio/min": 0.30277448892593384, "sampling/sampling_logp_difference/max": 1.1947669982910156, "sampling/sampling_logp_difference/mean": 0.014815653674304485, "step": 2905 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.002388683380559087, "epoch": 0.11672089006707635, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.196969696969697e-06, "loss": 0.0, "num_tokens": 6592674.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.003775715827942, "sampling/importance_sampling_ratio/mean": 1.0002539157867432, "sampling/importance_sampling_ratio/min": 0.9966413378715515, "sampling/sampling_logp_difference/max": 0.003768636379390955, "sampling/sampling_logp_difference/mean": 0.00027867371682077646, "step": 2906 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.001633810141356662, "epoch": 0.1167610555488613, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.193939393939394e-06, "loss": 0.0, "num_tokens": 6594410.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.002988338470459, "sampling/importance_sampling_ratio/mean": 1.0002257823944092, "sampling/importance_sampling_ratio/min": 0.9999564290046692, "sampling/sampling_logp_difference/max": 0.0029838387854397297, "sampling/sampling_logp_difference/mean": 0.00022611531312577426, "step": 2907 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.03689346136525273, "epoch": 0.11680122103064626, "frac_reward_zero_std": 0.0, "grad_norm": 0.9043018221855164, "learning_rate": 1.190909090909091e-06, "loss": -0.0005, "num_tokens": 6595818.0, "reward": 0.9995806217193604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9995806217193604, "reward_meter_std": 1.9758595954044722e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.9746988982660696e-05, "reward_total_composite_mean": 0.9995806217193604, "reward_total_composite_std": 1.9758595954044722e-05, "reward_total_mean": 0.9995806217193604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9995806217193604, "rewards/meter/std": 1.9758595954044722e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995806217193604, "rewards/total_composite/std": 1.9758595954044722e-05, "sampling/importance_sampling_ratio/max": 1.241074562072754, "sampling/importance_sampling_ratio/mean": 1.0006860494613647, "sampling/importance_sampling_ratio/min": 0.4804760813713074, "sampling/sampling_logp_difference/max": 0.7329778671264648, "sampling/sampling_logp_difference/mean": 0.005839377176016569, "step": 2908 }, { "clip_ratio/high_max": 0.032285032561048865, "clip_ratio/high_mean": 0.032285032561048865, "clip_ratio/low_mean": 0.0035211266949772835, "clip_ratio/low_min": 0.0035211266949772835, "clip_ratio/region_mean": 0.03580615925602615, "completions/clipped_ratio": 0.0, "completions/max_length": 319.0, "completions/max_terminated_length": 319.0, "completions/mean_length": 306.0, "completions/mean_terminated_length": 306.0, "completions/min_length": 284.0, "completions/min_terminated_length": 284.0, "entropy": 0.5155597105622292, "epoch": 0.11684138651243121, "frac_reward_zero_std": 0.0, "grad_norm": 1.9418765306472778, "learning_rate": 1.187878787878788e-06, "loss": -0.0212, "num_tokens": 6600002.0, "reward": 0.7390196323394775, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.887499988079071, "reward_count_adherence_std": 0.035355325788259506, "reward_meter_mean": 0.973264217376709, "reward_meter_std": 0.044348716735839844, "reward_repeat_penalty_mean": 0.962775707244873, "reward_repeat_penalty_std": 0.044049203395843506, "reward_std": 0.3020605444908142, "reward_total_composite_mean": 0.7390196323394775, "reward_total_composite_std": 0.3020605444908142, "reward_total_mean": 0.7390196323394775, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.887499988079071, "rewards/count_adherence/std": 0.035355325788259506, "rewards/meter/mean": 0.973264217376709, "rewards/meter/std": 0.044348716735839844, "rewards/repeat_penalty/mean": 0.962775707244873, "rewards/repeat_penalty/std": 0.044049203395843506, "rewards/total_composite/mean": 0.7390196323394775, "rewards/total_composite/std": 0.3020605444908142, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0099446773529053, "sampling/importance_sampling_ratio/min": 0.17252032458782196, "sampling/sampling_logp_difference/max": 1.7572402954101562, "sampling/sampling_logp_difference/mean": 0.048917196691036224, "step": 2909 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.02543548052199185, "epoch": 0.11688155199421617, "frac_reward_zero_std": 0.0, "grad_norm": 0.2488185316324234, "learning_rate": 1.184848484848485e-06, "loss": 0.0001, "num_tokens": 6601362.0, "reward": 0.9995900392532349, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9995900392532349, "reward_meter_std": 2.9798916330037173e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.9798916330037173e-06, "reward_total_composite_mean": 0.9995900392532349, "reward_total_composite_std": 2.9798916330037173e-06, "reward_total_mean": 0.9995900392532349, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9995900392532349, "rewards/meter/std": 2.9798916330037173e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995900392532349, "rewards/total_composite/std": 2.9798916330037173e-06, "sampling/importance_sampling_ratio/max": 1.1393342018127441, "sampling/importance_sampling_ratio/mean": 0.9999272227287292, "sampling/importance_sampling_ratio/min": 0.4592705965042114, "sampling/sampling_logp_difference/max": 0.7781157493591309, "sampling/sampling_logp_difference/mean": 0.0049346331506967545, "step": 2910 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.03822743846103549, "epoch": 0.11692171747600112, "frac_reward_zero_std": 0.0, "grad_norm": 0.16482257843017578, "learning_rate": 1.181818181818182e-06, "loss": -0.0001, "num_tokens": 6603186.0, "reward": 0.9981510639190674, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981510639190674, "reward_meter_std": 1.0189157364948187e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.0199873031524476e-05, "reward_total_composite_mean": 0.9981510639190674, "reward_total_composite_std": 1.0189157364948187e-05, "reward_total_mean": 0.9981510639190674, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981510639190674, "rewards/meter/std": 1.0189157364948187e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981510639190674, "rewards/total_composite/std": 1.0189157364948187e-05, "sampling/importance_sampling_ratio/max": 1.1907520294189453, "sampling/importance_sampling_ratio/mean": 0.9997050762176514, "sampling/importance_sampling_ratio/min": 0.530920684337616, "sampling/sampling_logp_difference/max": 0.6331427097320557, "sampling/sampling_logp_difference/mean": 0.005824689287692308, "step": 2911 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.0074360816506668925, "clip_ratio/low_min": 0.0074360816506668925, "clip_ratio/region_mean": 0.009301753249019384, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.375, "completions/mean_terminated_length": 67.375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.12552092969417572, "epoch": 0.11696188295778608, "frac_reward_zero_std": 0.0, "grad_norm": 1.688647747039795, "learning_rate": 1.1787878787878788e-06, "loss": -0.0024, "num_tokens": 6605005.0, "reward": 0.99924635887146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99924635887146, "reward_meter_std": 0.0001372320402879268, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001372358383378014, "reward_total_composite_mean": 0.99924635887146, "reward_total_composite_std": 0.0001372320402879268, "reward_total_mean": 0.99924635887146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99924635887146, "rewards/meter/std": 0.0001372320402879268, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99924635887146, "rewards/total_composite/std": 0.0001372320402879268, "sampling/importance_sampling_ratio/max": 1.6964365243911743, "sampling/importance_sampling_ratio/mean": 1.0067250728607178, "sampling/importance_sampling_ratio/min": 0.32733336091041565, "sampling/sampling_logp_difference/max": 1.1167762279510498, "sampling/sampling_logp_difference/mean": 0.01559353992342949, "step": 2912 }, { "clip_ratio/high_max": 0.024883363395929337, "clip_ratio/high_mean": 0.024883363395929337, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.024883363395929337, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 69.125, "completions/mean_terminated_length": 69.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23639377392828465, "epoch": 0.11700204843957103, "frac_reward_zero_std": 0.0, "grad_norm": 5.435182094573975, "learning_rate": 1.1757575757575759e-06, "loss": -0.0167, "num_tokens": 6606638.0, "reward": 0.9914849996566772, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9914849996566772, "reward_meter_std": 0.013738686218857765, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.013738670386373997, "reward_total_composite_mean": 0.9914849996566772, "reward_total_composite_std": 0.013738686218857765, "reward_total_mean": 0.9914849996566772, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9914849996566772, "rewards/meter/std": 0.013738686218857765, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9914849996566772, "rewards/total_composite/std": 0.013738686218857765, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0020140409469604, "sampling/importance_sampling_ratio/min": 0.3607243597507477, "sampling/sampling_logp_difference/max": 1.0196411609649658, "sampling/sampling_logp_difference/mean": 0.027403326705098152, "step": 2913 }, { "clip_ratio/high_max": 0.052266900427639484, "clip_ratio/high_mean": 0.052266900427639484, "clip_ratio/low_mean": 0.004125412553548813, "clip_ratio/low_min": 0.004125412553548813, "clip_ratio/region_mean": 0.0563923129811883, "completions/clipped_ratio": 0.0, "completions/max_length": 312.0, "completions/max_terminated_length": 312.0, "completions/mean_length": 302.875, "completions/mean_terminated_length": 302.875, "completions/min_length": 291.0, "completions/min_terminated_length": 291.0, "entropy": 0.6028463244438171, "epoch": 0.11704221392135598, "frac_reward_zero_std": 0.0, "grad_norm": 2.8395214080810547, "learning_rate": 1.172727272727273e-06, "loss": 0.004, "num_tokens": 6610549.0, "reward": 0.8662627935409546, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9837737083435059, "reward_meter_std": 0.04140230268239975, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_std": 0.3506163954734802, "reward_total_composite_mean": 0.8662627935409546, "reward_total_composite_std": 0.3506163954734802, "reward_total_mean": 0.8662627935409546, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9837737083435059, "rewards/meter/std": 0.04140230268239975, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.8662627935409546, "rewards/total_composite/std": 0.3506163954734802, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010536551475525, "sampling/importance_sampling_ratio/min": 0.1935638189315796, "sampling/sampling_logp_difference/max": 1.6421480178833008, "sampling/sampling_logp_difference/mean": 0.059933777898550034, "step": 2914 }, { "clip_ratio/high_max": 0.012379350140690804, "clip_ratio/high_mean": 0.012379350140690804, "clip_ratio/low_mean": 0.017094367765821517, "clip_ratio/low_min": 0.017094367765821517, "clip_ratio/region_mean": 0.02947371790651232, "completions/clipped_ratio": 0.0, "completions/max_length": 286.0, "completions/max_terminated_length": 286.0, "completions/mean_length": 264.75, "completions/mean_terminated_length": 264.75, "completions/min_length": 256.0, "completions/min_terminated_length": 256.0, "entropy": 0.3080044835805893, "epoch": 0.11708237940314094, "frac_reward_zero_std": 0.0, "grad_norm": 2.1781363487243652, "learning_rate": 1.1696969696969697e-06, "loss": 0.0239, "num_tokens": 6614339.0, "reward": 0.9504234790802002, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9970918893814087, "reward_meter_std": 0.005339814815670252, "reward_repeat_penalty_mean": 0.9676470756530762, "reward_repeat_penalty_std": 0.03468189761042595, "reward_std": 0.06568736582994461, "reward_total_composite_mean": 0.9504234790802002, "reward_total_composite_std": 0.06568735092878342, "reward_total_mean": 0.9504234790802002, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9970918893814087, "rewards/meter/std": 0.005339814815670252, "rewards/repeat_penalty/mean": 0.9676470756530762, "rewards/repeat_penalty/std": 0.03468189761042595, "rewards/total_composite/mean": 0.9504234790802002, "rewards/total_composite/std": 0.06568735092878342, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0066348314285278, "sampling/importance_sampling_ratio/min": 0.24354904890060425, "sampling/sampling_logp_difference/max": 1.4124369621276855, "sampling/sampling_logp_difference/mean": 0.033314015716314316, "step": 2915 }, { "clip_ratio/high_max": 0.020882656681351364, "clip_ratio/high_mean": 0.020882656681351364, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.02277659613173455, "completions/clipped_ratio": 0.0, "completions/max_length": 201.0, "completions/max_terminated_length": 201.0, "completions/mean_length": 198.0, "completions/mean_terminated_length": 198.0, "completions/min_length": 196.0, "completions/min_terminated_length": 196.0, "entropy": 0.3740917034447193, "epoch": 0.11712254488492589, "frac_reward_zero_std": 0.0, "grad_norm": 1.5669199228286743, "learning_rate": 1.1666666666666668e-06, "loss": -0.0014, "num_tokens": 6617355.0, "reward": 0.9989873170852661, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989873170852661, "reward_meter_std": 0.000616980018094182, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006169742555357516, "reward_total_composite_mean": 0.9989873170852661, "reward_total_composite_std": 0.000616980018094182, "reward_total_mean": 0.9989873170852661, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989873170852661, "rewards/meter/std": 0.000616980018094182, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989873170852661, "rewards/total_composite/std": 0.000616980018094182, "sampling/importance_sampling_ratio/max": 1.8285475969314575, "sampling/importance_sampling_ratio/mean": 1.0102366209030151, "sampling/importance_sampling_ratio/min": 0.33361366391181946, "sampling/sampling_logp_difference/max": 1.0977716445922852, "sampling/sampling_logp_difference/mean": 0.03752731904387474, "step": 2916 }, { "clip_ratio/high_max": 0.006749970489181578, "clip_ratio/high_mean": 0.006749970489181578, "clip_ratio/low_mean": 0.008138206438161433, "clip_ratio/low_min": 0.008138206438161433, "clip_ratio/region_mean": 0.014888176927343011, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 92.5, "completions/mean_terminated_length": 92.5, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.12023854069411755, "epoch": 0.11716271036671085, "frac_reward_zero_std": 0.0, "grad_norm": 2.548100233078003, "learning_rate": 1.1636363636363638e-06, "loss": -0.0003, "num_tokens": 6619535.0, "reward": 0.9972745180130005, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972745180130005, "reward_meter_std": 0.00045341101940721273, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00045341774239204824, "reward_total_composite_mean": 0.9972745180130005, "reward_total_composite_std": 0.00045341101940721273, "reward_total_mean": 0.9972745180130005, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972745180130005, "rewards/meter/std": 0.00045341101940721273, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972745180130005, "rewards/total_composite/std": 0.00045341101940721273, "sampling/importance_sampling_ratio/max": 1.420615792274475, "sampling/importance_sampling_ratio/mean": 1.0014476776123047, "sampling/importance_sampling_ratio/min": 0.3447898030281067, "sampling/sampling_logp_difference/max": 1.0648202896118164, "sampling/sampling_logp_difference/mean": 0.016107600182294846, "step": 2917 }, { "clip_ratio/high_max": 0.019217372057028115, "clip_ratio/high_mean": 0.019217372057028115, "clip_ratio/low_mean": 0.0024271844886243343, "clip_ratio/low_min": 0.0024271844886243343, "clip_ratio/region_mean": 0.02164455654565245, "completions/clipped_ratio": 0.0, "completions/max_length": 109.0, "completions/max_terminated_length": 109.0, "completions/mean_length": 103.875, "completions/mean_terminated_length": 103.875, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.2945725992321968, "epoch": 0.1172028758484958, "frac_reward_zero_std": 0.0, "grad_norm": 3.6516404151916504, "learning_rate": 1.1606060606060607e-06, "loss": 0.0044, "num_tokens": 6621518.0, "reward": 0.9887614250183105, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9887614250183105, "reward_meter_std": 0.016851743683218956, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01685173623263836, "reward_total_composite_mean": 0.9887614250183105, "reward_total_composite_std": 0.016851743683218956, "reward_total_mean": 0.9887614250183105, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9887614250183105, "rewards/meter/std": 0.016851743683218956, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9887614250183105, "rewards/total_composite/std": 0.016851743683218956, "sampling/importance_sampling_ratio/max": 1.9795823097229004, "sampling/importance_sampling_ratio/mean": 1.0085445642471313, "sampling/importance_sampling_ratio/min": 0.2346658855676651, "sampling/sampling_logp_difference/max": 1.4495925903320312, "sampling/sampling_logp_difference/mean": 0.038346562534570694, "step": 2918 }, { "clip_ratio/high_max": 0.003504672786220908, "clip_ratio/high_mean": 0.003504672786220908, "clip_ratio/low_mean": 0.004705960163846612, "clip_ratio/low_min": 0.004705960163846612, "clip_ratio/region_mean": 0.00821063295006752, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 106.75, "completions/mean_terminated_length": 106.75, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.14118101634085178, "epoch": 0.11724304133028075, "frac_reward_zero_std": 0.0, "grad_norm": 1.5092811584472656, "learning_rate": 1.1575757575757577e-06, "loss": -0.0035, "num_tokens": 6623884.0, "reward": 0.999142587184906, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999142587184906, "reward_meter_std": 0.0002281271299580112, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00022812940005678684, "reward_total_composite_mean": 0.999142587184906, "reward_total_composite_std": 0.0002281271299580112, "reward_total_mean": 0.999142587184906, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999142587184906, "rewards/meter/std": 0.0002281271299580112, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999142587184906, "rewards/total_composite/std": 0.0002281271299580112, "sampling/importance_sampling_ratio/max": 1.5648819208145142, "sampling/importance_sampling_ratio/mean": 1.0026243925094604, "sampling/importance_sampling_ratio/min": 0.3634822368621826, "sampling/sampling_logp_difference/max": 1.0120248794555664, "sampling/sampling_logp_difference/mean": 0.016602281481027603, "step": 2919 }, { "clip_ratio/high_max": 0.00352671486325562, "clip_ratio/high_mean": 0.00352671486325562, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/region_mean": 0.005885205464437604, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 106.5, "completions/mean_terminated_length": 106.5, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.13282244093716145, "epoch": 0.11728320681206571, "frac_reward_zero_std": 0.0, "grad_norm": 0.6831678152084351, "learning_rate": 1.1545454545454545e-06, "loss": 0.002, "num_tokens": 6626176.0, "reward": 0.9992160201072693, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992160201072693, "reward_meter_std": 5.9427493397379294e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.943301584920846e-05, "reward_total_composite_mean": 0.9992160201072693, "reward_total_composite_std": 5.9427493397379294e-05, "reward_total_mean": 0.9992160201072693, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992160201072693, "rewards/meter/std": 5.9427493397379294e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992160201072693, "rewards/total_composite/std": 5.9427493397379294e-05, "sampling/importance_sampling_ratio/max": 1.5806970596313477, "sampling/importance_sampling_ratio/mean": 1.0034005641937256, "sampling/importance_sampling_ratio/min": 0.5394703149795532, "sampling/sampling_logp_difference/max": 0.6171674728393555, "sampling/sampling_logp_difference/mean": 0.013828075490891933, "step": 2920 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0019735290406970307, "epoch": 0.11732337229385066, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.1515151515151516e-06, "loss": 0.0, "num_tokens": 6627968.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0063046216964722, "sampling/importance_sampling_ratio/mean": 1.0000739097595215, "sampling/importance_sampling_ratio/min": 0.9933332204818726, "sampling/sampling_logp_difference/max": 0.006689060479402542, "sampling/sampling_logp_difference/mean": 0.0001586712896823883, "step": 2921 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.0035714285913854837, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 34.125, "completions/mean_terminated_length": 34.125, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.03445210144855082, "epoch": 0.11736353777563562, "frac_reward_zero_std": 0.0, "grad_norm": 0.311759889125824, "learning_rate": 1.1484848484848486e-06, "loss": 0.0033, "num_tokens": 6629489.0, "reward": 0.9968487024307251, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9968487024307251, "reward_meter_std": 0.0025374931283295155, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0025374931283295155, "reward_total_composite_mean": 0.9968487024307251, "reward_total_composite_std": 0.0025374931283295155, "reward_total_mean": 0.9968487024307251, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9968487024307251, "rewards/meter/std": 0.0025374931283295155, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9968487024307251, "rewards/total_composite/std": 0.0025374931283295155, "sampling/importance_sampling_ratio/max": 1.059367060661316, "sampling/importance_sampling_ratio/mean": 1.0004445314407349, "sampling/importance_sampling_ratio/min": 0.4213464856147766, "sampling/sampling_logp_difference/max": 0.8642997741699219, "sampling/sampling_logp_difference/mean": 0.006284201052039862, "step": 2922 }, { "clip_ratio/high_max": 0.03803631942719221, "clip_ratio/high_mean": 0.03803631942719221, "clip_ratio/low_mean": 0.00808823574334383, "clip_ratio/low_min": 0.00808823574334383, "clip_ratio/region_mean": 0.04612455517053604, "completions/clipped_ratio": 0.0, "completions/max_length": 173.0, "completions/max_terminated_length": 173.0, "completions/mean_length": 168.0, "completions/mean_terminated_length": 168.0, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.4091906175017357, "epoch": 0.11740370325742057, "frac_reward_zero_std": 0.0, "grad_norm": 4.833311557769775, "learning_rate": 1.1454545454545457e-06, "loss": 0.0051, "num_tokens": 6632353.0, "reward": 0.9983302354812622, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9983302354812622, "reward_meter_std": 0.0015840368578210473, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0015840278938412666, "reward_total_composite_mean": 0.9983302354812622, "reward_total_composite_std": 0.0015840368578210473, "reward_total_mean": 0.9983302354812622, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9983302354812622, "rewards/meter/std": 0.0015840368578210473, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9983302354812622, "rewards/total_composite/std": 0.0015840368578210473, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0118147134780884, "sampling/importance_sampling_ratio/min": 0.3586997985839844, "sampling/sampling_logp_difference/max": 1.0252695083618164, "sampling/sampling_logp_difference/mean": 0.04776981472969055, "step": 2923 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.0020491802133619785, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.0171608105301857, "epoch": 0.11744386873920552, "frac_reward_zero_std": 0.0, "grad_norm": 0.0032307966612279415, "learning_rate": 1.1424242424242425e-06, "loss": -0.0, "num_tokens": 6634097.0, "reward": 0.9973385334014893, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973385334014893, "reward_meter_std": 1.0115243185282452e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.0115243185282452e-06, "reward_total_composite_mean": 0.9973385334014893, "reward_total_composite_std": 1.0115243185282452e-06, "reward_total_mean": 0.9973385334014893, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973385334014893, "rewards/meter/std": 1.0115243185282452e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973385334014893, "rewards/total_composite/std": 1.0115243185282452e-06, "sampling/importance_sampling_ratio/max": 1.2659525871276855, "sampling/importance_sampling_ratio/mean": 1.0016769170761108, "sampling/importance_sampling_ratio/min": 0.9778724312782288, "sampling/sampling_logp_difference/max": 0.2358248233795166, "sampling/sampling_logp_difference/mean": 0.0019246861338615417, "step": 2924 }, { "clip_ratio/high_max": 0.00530050863744691, "clip_ratio/high_mean": 0.00530050863744691, "clip_ratio/low_mean": 0.007904067635536194, "clip_ratio/low_min": 0.007904067635536194, "clip_ratio/region_mean": 0.013204576272983104, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 141.875, "completions/mean_terminated_length": 141.875, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.1869067270308733, "epoch": 0.11748403422099048, "frac_reward_zero_std": 0.0, "grad_norm": 1.0003553628921509, "learning_rate": 1.1393939393939395e-06, "loss": 0.0039, "num_tokens": 6636616.0, "reward": 0.999082088470459, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999082088470459, "reward_meter_std": 0.00011351890861988068, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011351749708410352, "reward_total_composite_mean": 0.999082088470459, "reward_total_composite_std": 0.00011351890861988068, "reward_total_mean": 0.999082088470459, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999082088470459, "rewards/meter/std": 0.00011351890861988068, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999082088470459, "rewards/total_composite/std": 0.00011351890861988068, "sampling/importance_sampling_ratio/max": 1.6239076852798462, "sampling/importance_sampling_ratio/mean": 1.0028555393218994, "sampling/importance_sampling_ratio/min": 0.24545511603355408, "sampling/sampling_logp_difference/max": 1.4046411514282227, "sampling/sampling_logp_difference/mean": 0.02242162451148033, "step": 2925 }, { "clip_ratio/high_max": 0.029786661034449935, "clip_ratio/high_mean": 0.029786661034449935, "clip_ratio/low_mean": 0.00918415142223239, "clip_ratio/low_min": 0.00918415142223239, "clip_ratio/region_mean": 0.038970812456682324, "completions/clipped_ratio": 0.0, "completions/max_length": 220.0, "completions/max_terminated_length": 220.0, "completions/mean_length": 208.0, "completions/mean_terminated_length": 208.0, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "entropy": 0.47782545536756516, "epoch": 0.11752419970277543, "frac_reward_zero_std": 0.0, "grad_norm": 3.0424153804779053, "learning_rate": 1.1363636363636364e-06, "loss": -0.0091, "num_tokens": 6639784.0, "reward": 0.7895516157150269, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8537713289260864, "reward_meter_std": 0.2598452866077423, "reward_repeat_penalty_mean": 0.9204545617103577, "reward_repeat_penalty_std": 0.0758657231926918, "reward_std": 0.25156062841415405, "reward_total_composite_mean": 0.7895516157150269, "reward_total_composite_std": 0.25156062841415405, "reward_total_mean": 0.7895516157150269, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8537713289260864, "rewards/meter/std": 0.2598452866077423, "rewards/repeat_penalty/mean": 0.9204545617103577, "rewards/repeat_penalty/std": 0.0758657231926918, "rewards/total_composite/mean": 0.7895516157150269, "rewards/total_composite/std": 0.25156062841415405, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.012603998184204, "sampling/importance_sampling_ratio/min": 0.22566285729408264, "sampling/sampling_logp_difference/max": 1.488713264465332, "sampling/sampling_logp_difference/mean": 0.046991463750600815, "step": 2926 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0003692922255140729, "epoch": 0.11756436518456038, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.1333333333333334e-06, "loss": 0.0, "num_tokens": 6641264.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0004537105560303, "sampling/importance_sampling_ratio/mean": 1.0000441074371338, "sampling/importance_sampling_ratio/min": 0.999942421913147, "sampling/sampling_logp_difference/max": 0.00045358005445450544, "sampling/sampling_logp_difference/mean": 4.449375410331413e-05, "step": 2927 }, { "clip_ratio/high_max": 0.007456094026565552, "clip_ratio/high_mean": 0.007456094026565552, "clip_ratio/low_mean": 0.012113503995351493, "clip_ratio/low_min": 0.012113503995351493, "clip_ratio/region_mean": 0.019569598021917045, "completions/clipped_ratio": 0.0, "completions/max_length": 239.0, "completions/max_terminated_length": 239.0, "completions/mean_length": 236.625, "completions/mean_terminated_length": 236.625, "completions/min_length": 233.0, "completions/min_terminated_length": 233.0, "entropy": 0.3537592738866806, "epoch": 0.11760453066634534, "frac_reward_zero_std": 0.0, "grad_norm": 1.2671390771865845, "learning_rate": 1.1303030303030305e-06, "loss": 0.003, "num_tokens": 6644813.0, "reward": 0.9991750121116638, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991750121116638, "reward_meter_std": 0.00013307879271451384, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00013307768676895648, "reward_total_composite_mean": 0.9991750121116638, "reward_total_composite_std": 0.00013307879271451384, "reward_total_mean": 0.9991750121116638, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991750121116638, "rewards/meter/std": 0.00013307879271451384, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991750121116638, "rewards/total_composite/std": 0.00013307879271451384, "sampling/importance_sampling_ratio/max": 1.8521274328231812, "sampling/importance_sampling_ratio/mean": 1.0087391138076782, "sampling/importance_sampling_ratio/min": 0.3410716950893402, "sampling/sampling_logp_difference/max": 1.075662612915039, "sampling/sampling_logp_difference/mean": 0.035225577652454376, "step": 2928 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.0018939394503831863, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.029220073018223047, "epoch": 0.11764469614813029, "frac_reward_zero_std": 0.0, "grad_norm": 0.040891580283641815, "learning_rate": 1.1272727272727275e-06, "loss": -0.0, "num_tokens": 6646628.0, "reward": 0.9981529116630554, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981529116630554, "reward_meter_std": 2.428059815429151e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.428059815429151e-06, "reward_total_composite_mean": 0.9981529116630554, "reward_total_composite_std": 2.428059815429151e-06, "reward_total_mean": 0.9981529116630554, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981529116630554, "rewards/meter/std": 2.428059815429151e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981529116630554, "rewards/total_composite/std": 2.428059815429151e-06, "sampling/importance_sampling_ratio/max": 1.0871691703796387, "sampling/importance_sampling_ratio/mean": 1.0010944604873657, "sampling/importance_sampling_ratio/min": 0.7372743487358093, "sampling/sampling_logp_difference/max": 0.3047952651977539, "sampling/sampling_logp_difference/mean": 0.0030419672839343548, "step": 2929 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0018656715983524919, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.03760868404060602, "epoch": 0.11768486162991525, "frac_reward_zero_std": 0.0, "grad_norm": 0.1988767832517624, "learning_rate": 1.1242424242424243e-06, "loss": -0.0005, "num_tokens": 6648339.0, "reward": 0.998143196105957, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998143196105957, "reward_meter_std": 1.6269057596218772e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.6278554539894685e-05, "reward_total_composite_mean": 0.998143196105957, "reward_total_composite_std": 1.6269057596218772e-05, "reward_total_mean": 0.998143196105957, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998143196105957, "rewards/meter/std": 1.6269057596218772e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998143196105957, "rewards/total_composite/std": 1.6269057596218772e-05, "sampling/importance_sampling_ratio/max": 1.115285038948059, "sampling/importance_sampling_ratio/mean": 0.9996000528335571, "sampling/importance_sampling_ratio/min": 0.44213125109672546, "sampling/sampling_logp_difference/max": 0.8161485195159912, "sampling/sampling_logp_difference/mean": 0.007081957999616861, "step": 2930 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.002932642324594781, "epoch": 0.1177250271117002, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.1212121212121214e-06, "loss": 0.0, "num_tokens": 6650299.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0083446502685547, "sampling/importance_sampling_ratio/mean": 0.9990981221199036, "sampling/importance_sampling_ratio/min": 0.18908332288265228, "sampling/sampling_logp_difference/max": 1.665567398071289, "sampling/sampling_logp_difference/mean": 0.0029783351346850395, "step": 2931 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.0006702992213831749, "epoch": 0.11776519259348515, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.1181818181818182e-06, "loss": 0.0, "num_tokens": 6651915.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.000840187072754, "sampling/importance_sampling_ratio/mean": 1.0000754594802856, "sampling/importance_sampling_ratio/min": 0.9997352361679077, "sampling/sampling_logp_difference/max": 0.0008399296784773469, "sampling/sampling_logp_difference/mean": 7.779466250212863e-05, "step": 2932 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.0018656715983524919, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.028828308917582035, "epoch": 0.11780535807527011, "frac_reward_zero_std": 0.0, "grad_norm": 0.009405174292623997, "learning_rate": 1.1151515151515153e-06, "loss": -0.0002, "num_tokens": 6653707.0, "reward": 0.9981515407562256, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981515407562256, "reward_meter_std": 1.501989686403249e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.5057863720358e-06, "reward_total_composite_mean": 0.9981515407562256, "reward_total_composite_std": 1.501989686403249e-06, "reward_total_mean": 0.9981515407562256, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981515407562256, "rewards/meter/std": 1.501989686403249e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981515407562256, "rewards/total_composite/std": 1.501989686403249e-06, "sampling/importance_sampling_ratio/max": 1.0989731550216675, "sampling/importance_sampling_ratio/mean": 1.002063512802124, "sampling/importance_sampling_ratio/min": 0.9517659544944763, "sampling/sampling_logp_difference/max": 0.09437625110149384, "sampling/sampling_logp_difference/mean": 0.0026725942734628916, "step": 2933 }, { "clip_ratio/high_max": 0.01744664926081896, "clip_ratio/high_mean": 0.01744664926081896, "clip_ratio/low_mean": 0.001623376621864736, "clip_ratio/low_min": 0.001623376621864736, "clip_ratio/region_mean": 0.019070025882683694, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 78.625, "completions/mean_terminated_length": 78.625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.18547690846025944, "epoch": 0.11784552355705506, "frac_reward_zero_std": 0.0, "grad_norm": 2.1487016677856445, "learning_rate": 1.112121212121212e-06, "loss": -0.0064, "num_tokens": 6655568.0, "reward": 0.9986746311187744, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986746311187744, "reward_meter_std": 0.0008188642095774412, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008188657811842859, "reward_total_composite_mean": 0.9986746311187744, "reward_total_composite_std": 0.0008188642095774412, "reward_total_mean": 0.9986746311187744, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986746311187744, "rewards/meter/std": 0.0008188642095774412, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9986746311187744, "rewards/total_composite/std": 0.0008188642095774412, "sampling/importance_sampling_ratio/max": 1.461742877960205, "sampling/importance_sampling_ratio/mean": 1.002895712852478, "sampling/importance_sampling_ratio/min": 0.43293848633766174, "sampling/sampling_logp_difference/max": 0.8371596336364746, "sampling/sampling_logp_difference/mean": 0.020405031740665436, "step": 2934 }, { "clip_ratio/high_max": 0.008094357093796134, "clip_ratio/high_mean": 0.008094357093796134, "clip_ratio/low_mean": 0.005390953738242388, "clip_ratio/low_min": 0.005390953738242388, "clip_ratio/region_mean": 0.013485310832038522, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 92.75, "completions/mean_terminated_length": 92.75, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.09037415590137243, "epoch": 0.11788568903884002, "frac_reward_zero_std": 0.0, "grad_norm": 0.6610411405563354, "learning_rate": 1.1090909090909093e-06, "loss": 0.0009, "num_tokens": 6657678.0, "reward": 0.9977371692657471, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977371692657471, "reward_meter_std": 7.424020441249013e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.424017530865967e-05, "reward_total_composite_mean": 0.9977371692657471, "reward_total_composite_std": 7.424020441249013e-05, "reward_total_mean": 0.9977371692657471, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977371692657471, "rewards/meter/std": 7.424020441249013e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977371692657471, "rewards/total_composite/std": 7.424020441249013e-05, "sampling/importance_sampling_ratio/max": 1.3887077569961548, "sampling/importance_sampling_ratio/mean": 1.0004804134368896, "sampling/importance_sampling_ratio/min": 0.48225268721580505, "sampling/sampling_logp_difference/max": 0.7292871475219727, "sampling/sampling_logp_difference/mean": 0.011522673070430756, "step": 2935 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.11792585452062497, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 1.1060606060606062e-06, "loss": 0.0, "num_tokens": 6659326.0, "reward": 0.6306551694869995, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6499999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992560148239136, "reward_meter_std": 9.200150816468522e-05, "reward_repeat_penalty_mean": 0.9709615111351013, "reward_repeat_penalty_std": 0.027279434725642204, "reward_std": 0.017706912010908127, "reward_total_composite_mean": 0.6306551694869995, "reward_total_composite_std": 0.01770690269768238, "reward_total_mean": 0.6306551694869995, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6499999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992560148239136, "rewards/meter/std": 9.200150816468522e-05, "rewards/repeat_penalty/mean": 0.9709615111351013, "rewards/repeat_penalty/std": 0.027279434725642204, "rewards/total_composite/mean": 0.6306551694869995, "rewards/total_composite/std": 0.01770690269768238, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 2936 }, { "clip_ratio/high_max": 0.006369685288518667, "clip_ratio/high_mean": 0.006369685288518667, "clip_ratio/low_mean": 0.007854048046283424, "clip_ratio/low_min": 0.007854048046283424, "clip_ratio/region_mean": 0.014223733334802091, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 79.0, "completions/mean_terminated_length": 79.0, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.16411693580448627, "epoch": 0.11796602000240992, "frac_reward_zero_std": 0.0, "grad_norm": 1.2551523447036743, "learning_rate": 1.1030303030303032e-06, "loss": 0.0054, "num_tokens": 6661254.0, "reward": 0.9990774989128113, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990774989128113, "reward_meter_std": 0.00010208567255176604, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010208544699708, "reward_total_composite_mean": 0.9990774989128113, "reward_total_composite_std": 0.00010208567255176604, "reward_total_mean": 0.9990774989128113, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990774989128113, "rewards/meter/std": 0.00010208567255176604, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990774989128113, "rewards/total_composite/std": 0.00010208567255176604, "sampling/importance_sampling_ratio/max": 1.4563788175582886, "sampling/importance_sampling_ratio/mean": 1.0067074298858643, "sampling/importance_sampling_ratio/min": 0.6219900250434875, "sampling/sampling_logp_difference/max": 0.474831223487854, "sampling/sampling_logp_difference/mean": 0.01773812435567379, "step": 2937 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0009229831339325756, "epoch": 0.11800618548419488, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.1e-06, "loss": 0.0, "num_tokens": 6662614.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.001272439956665, "sampling/importance_sampling_ratio/mean": 1.00003981590271, "sampling/importance_sampling_ratio/min": 0.9964448809623718, "sampling/sampling_logp_difference/max": 0.003561503253877163, "sampling/sampling_logp_difference/mean": 9.101704199565575e-05, "step": 2938 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.023612842429429293, "epoch": 0.11804635096597983, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.096969696969697e-06, "loss": 0.0, "num_tokens": 6664078.0, "reward": 0.9995916485786438, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9995916485786438, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9995916485786438, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9995916485786438, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9995916485786438, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995916485786438, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0143886804580688, "sampling/importance_sampling_ratio/mean": 1.001636266708374, "sampling/importance_sampling_ratio/min": 0.9820296168327332, "sampling/sampling_logp_difference/max": 0.01813378371298313, "sampling/sampling_logp_difference/mean": 0.002187325619161129, "step": 2939 }, { "clip_ratio/high_max": 0.009173946105875075, "clip_ratio/high_mean": 0.009173946105875075, "clip_ratio/low_mean": 0.007625087164342403, "clip_ratio/low_min": 0.007625087164342403, "clip_ratio/region_mean": 0.01679903327021748, "completions/clipped_ratio": 0.0, "completions/max_length": 304.0, "completions/max_terminated_length": 304.0, "completions/mean_length": 290.75, "completions/mean_terminated_length": 290.75, "completions/min_length": 284.0, "completions/min_terminated_length": 284.0, "entropy": 0.299512580037117, "epoch": 0.11808651644776479, "frac_reward_zero_std": 0.0, "grad_norm": 1.6798667907714844, "learning_rate": 1.093939393939394e-06, "loss": 0.0184, "num_tokens": 6668028.0, "reward": 0.9589698314666748, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990341663360596, "reward_meter_std": 9.884726023301482e-05, "reward_repeat_penalty_mean": 0.9598958492279053, "reward_repeat_penalty_std": 0.04713716357946396, "reward_std": 0.04711604863405228, "reward_total_composite_mean": 0.9589698314666748, "reward_total_composite_std": 0.04711604490876198, "reward_total_mean": 0.9589698314666748, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990341663360596, "rewards/meter/std": 9.884726023301482e-05, "rewards/repeat_penalty/mean": 0.9598958492279053, "rewards/repeat_penalty/std": 0.04713716357946396, "rewards/total_composite/mean": 0.9589698314666748, "rewards/total_composite/std": 0.04711604490876198, "sampling/importance_sampling_ratio/max": 1.720551609992981, "sampling/importance_sampling_ratio/mean": 1.0048670768737793, "sampling/importance_sampling_ratio/min": 0.19072523713111877, "sampling/sampling_logp_difference/max": 1.65692138671875, "sampling/sampling_logp_difference/mean": 0.026417069137096405, "step": 2940 }, { "clip_ratio/high_max": 0.009821378625929356, "clip_ratio/high_mean": 0.009821378625929356, "clip_ratio/low_mean": 0.003765060333535075, "clip_ratio/low_min": 0.003765060333535075, "clip_ratio/region_mean": 0.01358643895946443, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 165.75, "completions/mean_terminated_length": 165.75, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.11971556767821312, "epoch": 0.11812668192954974, "frac_reward_zero_std": 0.0, "grad_norm": 1.4318662881851196, "learning_rate": 1.090909090909091e-06, "loss": 0.0035, "num_tokens": 6670786.0, "reward": 0.9715272188186646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992879629135132, "reward_meter_std": 0.0003045983612537384, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05134394019842148, "reward_total_composite_mean": 0.9715272188186646, "reward_total_composite_std": 0.05134394392371178, "reward_total_mean": 0.9715272188186646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992879629135132, "rewards/meter/std": 0.0003045983612537384, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9715272188186646, "rewards/total_composite/std": 0.05134394392371178, "sampling/importance_sampling_ratio/max": 1.513466238975525, "sampling/importance_sampling_ratio/mean": 1.0033372640609741, "sampling/importance_sampling_ratio/min": 0.3099811375141144, "sampling/sampling_logp_difference/max": 1.1712437868118286, "sampling/sampling_logp_difference/mean": 0.015097087249159813, "step": 2941 }, { "clip_ratio/high_max": 0.026906014885753393, "clip_ratio/high_mean": 0.026906014885753393, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/region_mean": 0.029623406240716577, "completions/clipped_ratio": 0.0, "completions/max_length": 282.0, "completions/max_terminated_length": 282.0, "completions/mean_length": 278.875, "completions/mean_terminated_length": 278.875, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "entropy": 0.31575205735862255, "epoch": 0.1181668474113347, "frac_reward_zero_std": 0.0, "grad_norm": 2.797619581222534, "learning_rate": 1.087878787878788e-06, "loss": -0.0003, "num_tokens": 6674569.0, "reward": 0.8150506019592285, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974960684776306, "reward_meter_std": 0.000377756921807304, "reward_repeat_penalty_mean": 0.9338235855102539, "reward_repeat_penalty_std": 0.020797256380319595, "reward_std": 0.018201671540737152, "reward_total_composite_mean": 0.8150506019592285, "reward_total_composite_std": 0.018201656639575958, "reward_total_mean": 0.8150506019592285, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974960684776306, "rewards/meter/std": 0.000377756921807304, "rewards/repeat_penalty/mean": 0.9338235855102539, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.8150506019592285, "rewards/total_composite/std": 0.018201656639575958, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0059531927108765, "sampling/importance_sampling_ratio/min": 0.22986207902431488, "sampling/sampling_logp_difference/max": 1.47027587890625, "sampling/sampling_logp_difference/mean": 0.03948317840695381, "step": 2942 }, { "clip_ratio/high_max": 0.006298449821770191, "clip_ratio/high_mean": 0.006298449821770191, "clip_ratio/low_mean": 0.02794346760492772, "clip_ratio/low_min": 0.02794346760492772, "clip_ratio/region_mean": 0.03424191742669791, "completions/clipped_ratio": 0.0, "completions/max_length": 284.0, "completions/max_terminated_length": 284.0, "completions/mean_length": 271.5, "completions/mean_terminated_length": 271.5, "completions/min_length": 258.0, "completions/min_terminated_length": 258.0, "entropy": 0.37596092373132706, "epoch": 0.11820701289311965, "frac_reward_zero_std": 0.0, "grad_norm": 3.1147727966308594, "learning_rate": 1.084848484848485e-06, "loss": 0.0233, "num_tokens": 6678237.0, "reward": 0.8254706859588623, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9941989779472351, "reward_meter_std": 0.0028337803669273853, "reward_repeat_penalty_mean": 0.9489378929138184, "reward_repeat_penalty_std": 0.02066386677324772, "reward_std": 0.0162269975990057, "reward_total_composite_mean": 0.8254706859588623, "reward_total_composite_std": 0.016226978972554207, "reward_total_mean": 0.8254706859588623, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9941989779472351, "rewards/meter/std": 0.0028337803669273853, "rewards/repeat_penalty/mean": 0.9489378929138184, "rewards/repeat_penalty/std": 0.02066386677324772, "rewards/total_composite/mean": 0.8254706859588623, "rewards/total_composite/std": 0.016226978972554207, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0050019025802612, "sampling/importance_sampling_ratio/min": 0.02029639668762684, "sampling/sampling_logp_difference/max": 3.8973119258880615, "sampling/sampling_logp_difference/mean": 0.049873415380716324, "step": 2943 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.125, "completions/mean_terminated_length": 67.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.027955329976975918, "epoch": 0.1182471783749046, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.081818181818182e-06, "loss": 0.0, "num_tokens": 6680078.0, "reward": 0.9981522560119629, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981522560119629, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9981522560119629, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9981522560119629, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981522560119629, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981522560119629, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.359727144241333, "sampling/importance_sampling_ratio/mean": 1.0029489994049072, "sampling/importance_sampling_ratio/min": 0.946239173412323, "sampling/sampling_logp_difference/max": 0.3072841167449951, "sampling/sampling_logp_difference/mean": 0.0032630774658173323, "step": 2944 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.01105684821959585, "clip_ratio/low_min": 0.01105684821959585, "clip_ratio/region_mean": 0.012895083520561457, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.08658721391111612, "epoch": 0.11828734385668956, "frac_reward_zero_std": 0.0, "grad_norm": 1.8361525535583496, "learning_rate": 1.078787878787879e-06, "loss": -0.0019, "num_tokens": 6681590.0, "reward": 0.9993088841438293, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993088841438293, "reward_meter_std": 0.0001494528987677768, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001494480820838362, "reward_total_composite_mean": 0.9993088841438293, "reward_total_composite_std": 0.0001494528987677768, "reward_total_mean": 0.9993088841438293, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993088841438293, "rewards/meter/std": 0.0001494528987677768, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993088841438293, "rewards/total_composite/std": 0.0001494528987677768, "sampling/importance_sampling_ratio/max": 1.2217551469802856, "sampling/importance_sampling_ratio/mean": 1.0019557476043701, "sampling/importance_sampling_ratio/min": 0.5084711313247681, "sampling/sampling_logp_difference/max": 0.6763467788696289, "sampling/sampling_logp_difference/mean": 0.01226960588246584, "step": 2945 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0023213086824398488, "epoch": 0.11832750933847451, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.0757575757575758e-06, "loss": 0.0, "num_tokens": 6683270.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0045647621154785, "sampling/importance_sampling_ratio/mean": 1.0003174543380737, "sampling/importance_sampling_ratio/min": 0.9995915293693542, "sampling/sampling_logp_difference/max": 0.004554374143481255, "sampling/sampling_logp_difference/mean": 0.0003203663509339094, "step": 2946 }, { "clip_ratio/high_max": 0.015964585822075605, "clip_ratio/high_mean": 0.015964585822075605, "clip_ratio/low_mean": 0.02329266769811511, "clip_ratio/low_min": 0.02329266769811511, "clip_ratio/region_mean": 0.039257253520190716, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 118.875, "completions/mean_terminated_length": 118.875, "completions/min_length": 115.0, "completions/min_terminated_length": 115.0, "entropy": 0.20067131519317627, "epoch": 0.11836767482025946, "frac_reward_zero_std": 0.0, "grad_norm": 3.6917295455932617, "learning_rate": 1.0727272727272728e-06, "loss": 0.0031, "num_tokens": 6685877.0, "reward": 0.9246125221252441, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956615567207336, "reward_meter_std": 0.0021291025914251804, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.07701455056667328, "reward_total_composite_mean": 0.9246125221252441, "reward_total_composite_std": 0.07701455056667328, "reward_total_mean": 0.9246125221252441, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956615567207336, "rewards/meter/std": 0.0021291025914251804, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.9246125221252441, "rewards/total_composite/std": 0.07701455056667328, "sampling/importance_sampling_ratio/max": 1.86605966091156, "sampling/importance_sampling_ratio/mean": 1.0035653114318848, "sampling/importance_sampling_ratio/min": 0.20692089200019836, "sampling/sampling_logp_difference/max": 1.5754187107086182, "sampling/sampling_logp_difference/mean": 0.03363678976893425, "step": 2947 }, { "clip_ratio/high_max": 0.01954207243397832, "clip_ratio/high_mean": 0.01954207243397832, "clip_ratio/low_mean": 0.007194617064669728, "clip_ratio/low_min": 0.007194617064669728, "clip_ratio/region_mean": 0.026736689498648047, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 69.25, "completions/mean_terminated_length": 69.25, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.31914436258375645, "epoch": 0.11840784030204442, "frac_reward_zero_std": 0.0, "grad_norm": 3.9713573455810547, "learning_rate": 1.0696969696969696e-06, "loss": 0.0056, "num_tokens": 6687719.0, "reward": 0.9902843236923218, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9902843236923218, "reward_meter_std": 0.01087514590471983, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010875147767364979, "reward_total_composite_mean": 0.9902843236923218, "reward_total_composite_std": 0.01087514590471983, "reward_total_mean": 0.9902843236923218, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9902843236923218, "rewards/meter/std": 0.01087514590471983, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9902843236923218, "rewards/total_composite/std": 0.01087514590471983, "sampling/importance_sampling_ratio/max": 1.6766802072525024, "sampling/importance_sampling_ratio/mean": 0.9977670311927795, "sampling/importance_sampling_ratio/min": 0.16816474497318268, "sampling/sampling_logp_difference/max": 1.782811164855957, "sampling/sampling_logp_difference/mean": 0.040240850299596786, "step": 2948 }, { "clip_ratio/high_max": 0.01359617244452238, "clip_ratio/high_mean": 0.01359617244452238, "clip_ratio/low_mean": 0.0019379844889044762, "clip_ratio/low_min": 0.0019379844889044762, "clip_ratio/region_mean": 0.015534156933426857, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 128.75, "completions/mean_terminated_length": 128.75, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.15123706310987473, "epoch": 0.11844800578382937, "frac_reward_zero_std": 0.0, "grad_norm": 2.246262311935425, "learning_rate": 1.066666666666667e-06, "loss": -0.001, "num_tokens": 6690261.0, "reward": 0.9919617176055908, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9919617176055908, "reward_meter_std": 0.01668241061270237, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01668240688741207, "reward_total_composite_mean": 0.9919617176055908, "reward_total_composite_std": 0.01668241061270237, "reward_total_mean": 0.9919617176055908, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9919617176055908, "rewards/meter/std": 0.01668241061270237, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9919617176055908, "rewards/total_composite/std": 0.01668241061270237, "sampling/importance_sampling_ratio/max": 1.6150113344192505, "sampling/importance_sampling_ratio/mean": 0.999458909034729, "sampling/importance_sampling_ratio/min": 0.24293164908885956, "sampling/sampling_logp_difference/max": 1.4149751663208008, "sampling/sampling_logp_difference/mean": 0.017122527584433556, "step": 2949 }, { "clip_ratio/high_max": 0.009542036801576614, "clip_ratio/high_mean": 0.009542036801576614, "clip_ratio/low_mean": 0.009927221923135221, "clip_ratio/low_min": 0.009927221923135221, "clip_ratio/region_mean": 0.019469258724711835, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 90.125, "completions/mean_terminated_length": 90.125, "completions/min_length": 88.0, "completions/min_terminated_length": 88.0, "entropy": 0.16245986334979534, "epoch": 0.11848817126561433, "frac_reward_zero_std": 0.0, "grad_norm": 4.775022029876709, "learning_rate": 1.0636363636363637e-06, "loss": -0.0107, "num_tokens": 6692334.0, "reward": 0.8962882161140442, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957476258277893, "reward_meter_std": 0.0016542106168344617, "reward_repeat_penalty_mean": 0.8999999761581421, "reward_repeat_penalty_std": 0.10690449178218842, "reward_std": 0.107563816010952, "reward_total_composite_mean": 0.8962882161140442, "reward_total_composite_std": 0.10756382346153259, "reward_total_mean": 0.8962882161140442, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957476258277893, "rewards/meter/std": 0.0016542106168344617, "rewards/repeat_penalty/mean": 0.8999999761581421, "rewards/repeat_penalty/std": 0.10690449178218842, "rewards/total_composite/mean": 0.8962882161140442, "rewards/total_composite/std": 0.10756382346153259, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9992499947547913, "sampling/importance_sampling_ratio/min": 0.03590155765414238, "sampling/sampling_logp_difference/max": 3.326974630355835, "sampling/sampling_logp_difference/mean": 0.03922773897647858, "step": 2950 }, { "epoch": 0.11848817126561433, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/max_length": 403.61538461538464, "eval_completions/max_terminated_length": 395.0769230769231, "eval_completions/mean_length": 208.25, "eval_completions/mean_terminated_length": 205.135989849384, "eval_completions/min_length": 60.15384615384615, "eval_completions/min_terminated_length": 60.15384615384615, "eval_entropy": 0.38887230020302993, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 6692334.0, "eval_reward": 0.7131056900207813, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.06280488005051246, "eval_reward_count_adherence_mean": 0.9516645761636587, "eval_reward_count_adherence_std": 0.06901963714223641, "eval_reward_meter_mean": 0.7935946950545678, "eval_reward_meter_std": 0.32434708754030556, "eval_reward_repeat_penalty_mean": 0.9551343642748319, "eval_reward_repeat_penalty_std": 0.08046436768311721, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7131056900207813, "eval_reward_total_composite_std": 0.3318366717833739, "eval_reward_total_mean": 0.7131056900207813, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.06280488005051246, "eval_rewards/count_adherence/mean": 0.9516645761636587, "eval_rewards/count_adherence/std": 0.06901963714223641, "eval_rewards/meter/mean": 0.7935946950545678, "eval_rewards/meter/std": 0.32434708754030556, "eval_rewards/repeat_penalty/mean": 0.9551343642748319, "eval_rewards/repeat_penalty/std": 0.08046436768311721, "eval_rewards/total_composite/mean": 0.7131056900207813, "eval_rewards/total_composite/std": 0.3318366717833739, "eval_runtime": 75.4625, "eval_samples_per_second": 1.378, "eval_sampling/importance_sampling_ratio/max": 1.446980045391963, "eval_sampling/importance_sampling_ratio/mean": 1.0077741604584913, "eval_sampling/importance_sampling_ratio/min": 0.27521977172448087, "eval_sampling/sampling_logp_difference/max": 1.3049519062042236, "eval_sampling/sampling_logp_difference/mean": 0.032634207692283854, "eval_steps_per_second": 0.172, "step": 2950 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.0037878789007663727, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.625, "completions/mean_terminated_length": 33.625, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.1007662764750421, "epoch": 0.11852833674739928, "frac_reward_zero_std": 0.0, "grad_norm": 7.092353820800781, "learning_rate": 1.0606060606060608e-06, "loss": -0.0003, "num_tokens": 6693891.0, "reward": 0.9904308915138245, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9904308915138245, "reward_meter_std": 0.004542448092252016, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004542447626590729, "reward_total_composite_mean": 0.9904308915138245, "reward_total_composite_std": 0.004542448092252016, "reward_total_mean": 0.9904308915138245, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9904308915138245, "rewards/meter/std": 0.004542448092252016, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9904308915138245, "rewards/total_composite/std": 0.004542448092252016, "sampling/importance_sampling_ratio/max": 1.398740530014038, "sampling/importance_sampling_ratio/mean": 1.0013483762741089, "sampling/importance_sampling_ratio/min": 0.27674299478530884, "sampling/sampling_logp_difference/max": 1.2846660614013672, "sampling/sampling_logp_difference/mean": 0.021938247606158257, "step": 2951 }, { "clip_ratio/high_max": 0.013795045204460621, "clip_ratio/high_mean": 0.013795045204460621, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.013795045204460621, "completions/clipped_ratio": 0.0, "completions/max_length": 38.0, "completions/max_terminated_length": 38.0, "completions/mean_length": 36.5, "completions/mean_terminated_length": 36.5, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.03357855952344835, "epoch": 0.11856850222918423, "frac_reward_zero_std": 0.0, "grad_norm": 0.5699337124824524, "learning_rate": 1.0575757575757576e-06, "loss": 0.0018, "num_tokens": 6695487.0, "reward": 0.9995974898338318, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9995974898338318, "reward_meter_std": 1.9893313947250135e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.9893312128260732e-05, "reward_total_composite_mean": 0.9995974898338318, "reward_total_composite_std": 1.9893313947250135e-05, "reward_total_mean": 0.9995974898338318, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9995974898338318, "rewards/meter/std": 1.9893313947250135e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9995974898338318, "rewards/total_composite/std": 1.9893313947250135e-05, "sampling/importance_sampling_ratio/max": 1.0766030550003052, "sampling/importance_sampling_ratio/mean": 0.9967590570449829, "sampling/importance_sampling_ratio/min": 0.42069464921951294, "sampling/sampling_logp_difference/max": 0.8658480644226074, "sampling/sampling_logp_difference/mean": 0.009155918844044209, "step": 2952 }, { "clip_ratio/high_max": 0.00930431904271245, "clip_ratio/high_mean": 0.00930431904271245, "clip_ratio/low_mean": 0.002336448524147272, "clip_ratio/low_min": 0.002336448524147272, "clip_ratio/region_mean": 0.011640767566859722, "completions/clipped_ratio": 0.0, "completions/max_length": 110.0, "completions/max_terminated_length": 110.0, "completions/mean_length": 107.25, "completions/mean_terminated_length": 107.25, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.13113553449511528, "epoch": 0.11860866771096919, "frac_reward_zero_std": 0.0, "grad_norm": 1.022800087928772, "learning_rate": 1.0545454545454547e-06, "loss": -0.0001, "num_tokens": 6697761.0, "reward": 0.9992892742156982, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992892742156982, "reward_meter_std": 9.131882688961923e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.13253243197687e-05, "reward_total_composite_mean": 0.9992892742156982, "reward_total_composite_std": 9.131882688961923e-05, "reward_total_mean": 0.9992892742156982, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992892742156982, "rewards/meter/std": 9.131882688961923e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992892742156982, "rewards/total_composite/std": 9.131882688961923e-05, "sampling/importance_sampling_ratio/max": 1.6504535675048828, "sampling/importance_sampling_ratio/mean": 1.0017812252044678, "sampling/importance_sampling_ratio/min": 0.3837863802909851, "sampling/sampling_logp_difference/max": 0.9576692581176758, "sampling/sampling_logp_difference/mean": 0.016012487933039665, "step": 2953 }, { "clip_ratio/high_max": 0.04968656366690993, "clip_ratio/high_mean": 0.04968656366690993, "clip_ratio/low_mean": 0.024695121683180332, "clip_ratio/low_min": 0.024695121683180332, "clip_ratio/region_mean": 0.07438168535009027, "completions/clipped_ratio": 0.0, "completions/max_length": 45.0, "completions/max_terminated_length": 45.0, "completions/mean_length": 42.125, "completions/mean_terminated_length": 42.125, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.30831943079829216, "epoch": 0.11864883319275414, "frac_reward_zero_std": 0.0, "grad_norm": 10.395458221435547, "learning_rate": 1.0515151515151515e-06, "loss": -0.0034, "num_tokens": 6699386.0, "reward": 0.9281213283538818, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9281213283538818, "reward_meter_std": 0.031521886587142944, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03152187913656235, "reward_total_composite_mean": 0.9281213283538818, "reward_total_composite_std": 0.031521886587142944, "reward_total_mean": 0.9281213283538818, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9281213283538818, "rewards/meter/std": 0.031521886587142944, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9281213283538818, "rewards/total_composite/std": 0.031521886587142944, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9940267205238342, "sampling/importance_sampling_ratio/min": 0.060635656118392944, "sampling/sampling_logp_difference/max": 2.8028721809387207, "sampling/sampling_logp_difference/mean": 0.08637839555740356, "step": 2954 }, { "clip_ratio/high_max": 0.022244701394811273, "clip_ratio/high_mean": 0.022244701394811273, "clip_ratio/low_mean": 0.00557116128038615, "clip_ratio/low_min": 0.00557116128038615, "clip_ratio/region_mean": 0.027815862675197423, "completions/clipped_ratio": 0.0, "completions/max_length": 180.0, "completions/max_terminated_length": 180.0, "completions/mean_length": 174.0, "completions/mean_terminated_length": 174.0, "completions/min_length": 166.0, "completions/min_terminated_length": 166.0, "entropy": 0.3894418552517891, "epoch": 0.1186889986745391, "frac_reward_zero_std": 0.0, "grad_norm": 3.04614520072937, "learning_rate": 1.0484848484848485e-06, "loss": 0.0177, "num_tokens": 6702346.0, "reward": 0.9902915358543396, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9902915358543396, "reward_meter_std": 0.007541172672063112, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.007541188970208168, "reward_total_composite_mean": 0.9902915358543396, "reward_total_composite_std": 0.007541172672063112, "reward_total_mean": 0.9902915358543396, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9902915358543396, "rewards/meter/std": 0.007541172672063112, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9902915358543396, "rewards/total_composite/std": 0.007541172672063112, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0090171098709106, "sampling/importance_sampling_ratio/min": 0.25400206446647644, "sampling/sampling_logp_difference/max": 1.370412826538086, "sampling/sampling_logp_difference/mean": 0.04366350546479225, "step": 2955 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.007299659075215459, "clip_ratio/low_min": 0.007299659075215459, "clip_ratio/region_mean": 0.009137894376181066, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.0865520965307951, "epoch": 0.11872916415632405, "frac_reward_zero_std": 0.0, "grad_norm": 2.19303035736084, "learning_rate": 1.0454545454545456e-06, "loss": 0.0014, "num_tokens": 6704123.0, "reward": 0.9994041919708252, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994041919708252, "reward_meter_std": 0.0001539249497000128, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001539262884762138, "reward_total_composite_mean": 0.9994041919708252, "reward_total_composite_std": 0.0001539249497000128, "reward_total_mean": 0.9994041919708252, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994041919708252, "rewards/meter/std": 0.0001539249497000128, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994041919708252, "rewards/total_composite/std": 0.0001539249497000128, "sampling/importance_sampling_ratio/max": 1.366014838218689, "sampling/importance_sampling_ratio/mean": 1.0002238750457764, "sampling/importance_sampling_ratio/min": 0.35505202412605286, "sampling/sampling_logp_difference/max": 1.0354909896850586, "sampling/sampling_logp_difference/mean": 0.012835102155804634, "step": 2956 }, { "clip_ratio/high_max": 0.0056535504991188645, "clip_ratio/high_mean": 0.0056535504991188645, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0056535504991188645, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.04649371886625886, "epoch": 0.118769329638109, "frac_reward_zero_std": 0.0, "grad_norm": 0.0930638462305069, "learning_rate": 1.0424242424242426e-06, "loss": 0.0006, "num_tokens": 6705889.0, "reward": 0.998155951499939, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998155951499939, "reward_meter_std": 9.95435857475968e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.94737092696596e-06, "reward_total_composite_mean": 0.998155951499939, "reward_total_composite_std": 9.95435857475968e-06, "reward_total_mean": 0.998155951499939, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998155951499939, "rewards/meter/std": 9.95435857475968e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998155951499939, "rewards/total_composite/std": 9.95435857475968e-06, "sampling/importance_sampling_ratio/max": 1.253778100013733, "sampling/importance_sampling_ratio/mean": 0.9991642832756042, "sampling/importance_sampling_ratio/min": 0.44049713015556335, "sampling/sampling_logp_difference/max": 0.8198513984680176, "sampling/sampling_logp_difference/mean": 0.008290099911391735, "step": 2957 }, { "clip_ratio/high_max": 0.023225491400808096, "clip_ratio/high_mean": 0.023225491400808096, "clip_ratio/low_mean": 0.016178023535758257, "clip_ratio/low_min": 0.016178023535758257, "clip_ratio/region_mean": 0.03940351493656635, "completions/clipped_ratio": 0.0, "completions/max_length": 348.0, "completions/max_terminated_length": 348.0, "completions/mean_length": 319.875, "completions/mean_terminated_length": 319.875, "completions/min_length": 293.0, "completions/min_terminated_length": 293.0, "entropy": 0.47900502756237984, "epoch": 0.11880949511989396, "frac_reward_zero_std": 0.0, "grad_norm": 2.60581636428833, "learning_rate": 1.0393939393939394e-06, "loss": -0.0383, "num_tokens": 6710000.0, "reward": 0.9288632869720459, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.949999988079071, "reward_count_adherence_std": 0.0534522607922554, "reward_meter_mean": 0.9985268712043762, "reward_meter_std": 0.0011142482981085777, "reward_repeat_penalty_mean": 0.9794891476631165, "reward_repeat_penalty_std": 0.028372056782245636, "reward_std": 0.05354484170675278, "reward_total_composite_mean": 0.9288632869720459, "reward_total_composite_std": 0.05354485660791397, "reward_total_mean": 0.9288632869720459, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.949999988079071, "rewards/count_adherence/std": 0.0534522607922554, "rewards/meter/mean": 0.9985268712043762, "rewards/meter/std": 0.0011142482981085777, "rewards/repeat_penalty/mean": 0.9794891476631165, "rewards/repeat_penalty/std": 0.028372056782245636, "rewards/total_composite/mean": 0.9288632869720459, "rewards/total_composite/std": 0.05354485660791397, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0072740316390991, "sampling/importance_sampling_ratio/min": 0.14223718643188477, "sampling/sampling_logp_difference/max": 1.9502592086791992, "sampling/sampling_logp_difference/mean": 0.0562576949596405, "step": 2958 }, { "clip_ratio/high_max": 0.022940220427699387, "clip_ratio/high_mean": 0.022940220427699387, "clip_ratio/low_mean": 0.004202505107969046, "clip_ratio/low_min": 0.004202505107969046, "clip_ratio/region_mean": 0.027142725535668433, "completions/clipped_ratio": 0.0, "completions/max_length": 382.0, "completions/max_terminated_length": 382.0, "completions/mean_length": 358.125, "completions/mean_terminated_length": 358.125, "completions/min_length": 348.0, "completions/min_terminated_length": 348.0, "entropy": 0.40936222299933434, "epoch": 0.11884966060167891, "frac_reward_zero_std": 0.0, "grad_norm": 1.92487370967865, "learning_rate": 1.0363636363636365e-06, "loss": -0.0149, "num_tokens": 6714601.0, "reward": 0.8171521425247192, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8295454978942871, "reward_count_adherence_std": 0.03214123100042343, "reward_meter_mean": 0.9991495013237, "reward_meter_std": 0.0001215613738168031, "reward_repeat_penalty_mean": 0.985702633857727, "reward_repeat_penalty_std": 0.02648802287876606, "reward_std": 0.042467836290597916, "reward_total_composite_mean": 0.8171521425247192, "reward_total_composite_std": 0.04246782884001732, "reward_total_mean": 0.8171521425247192, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8295454978942871, "rewards/count_adherence/std": 0.03214123100042343, "rewards/meter/mean": 0.9991495013237, "rewards/meter/std": 0.0001215613738168031, "rewards/repeat_penalty/mean": 0.985702633857727, "rewards/repeat_penalty/std": 0.02648802287876606, "rewards/total_composite/mean": 0.8171521425247192, "rewards/total_composite/std": 0.04246782884001732, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0104491710662842, "sampling/importance_sampling_ratio/min": 0.228327676653862, "sampling/sampling_logp_difference/max": 1.476973533630371, "sampling/sampling_logp_difference/mean": 0.04391275718808174, "step": 2959 }, { "clip_ratio/high_max": 0.004899043240584433, "clip_ratio/high_mean": 0.004899043240584433, "clip_ratio/low_mean": 0.009166883770376444, "clip_ratio/low_min": 0.009166883770376444, "clip_ratio/region_mean": 0.014065927010960877, "completions/clipped_ratio": 0.0, "completions/max_length": 360.0, "completions/max_terminated_length": 360.0, "completions/mean_length": 355.125, "completions/mean_terminated_length": 355.125, "completions/min_length": 340.0, "completions/min_terminated_length": 340.0, "entropy": 0.3090662658214569, "epoch": 0.11888982608346386, "frac_reward_zero_std": 0.0, "grad_norm": 2.4538395404815674, "learning_rate": 1.0333333333333333e-06, "loss": -0.0038, "num_tokens": 6719098.0, "reward": 0.8728748559951782, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8977272510528564, "reward_count_adherence_std": 0.03214123100042343, "reward_meter_mean": 0.9989354610443115, "reward_meter_std": 0.0001577387156430632, "reward_repeat_penalty_mean": 0.9736841917037964, "reward_repeat_penalty_std": 0.02813275158405304, "reward_std": 0.032649945467710495, "reward_total_composite_mean": 0.8728748559951782, "reward_total_composite_std": 0.03264994919300079, "reward_total_mean": 0.8728748559951782, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8977272510528564, "rewards/count_adherence/std": 0.03214123100042343, "rewards/meter/mean": 0.9989354610443115, "rewards/meter/std": 0.0001577387156430632, "rewards/repeat_penalty/mean": 0.9736841917037964, "rewards/repeat_penalty/std": 0.02813275158405304, "rewards/total_composite/mean": 0.8728748559951782, "rewards/total_composite/std": 0.03264994919300079, "sampling/importance_sampling_ratio/max": 1.6595386266708374, "sampling/importance_sampling_ratio/mean": 1.0067881345748901, "sampling/importance_sampling_ratio/min": 0.29753369092941284, "sampling/sampling_logp_difference/max": 1.2122278213500977, "sampling/sampling_logp_difference/mean": 0.022427460178732872, "step": 2960 }, { "clip_ratio/high_max": 0.0521149430423975, "clip_ratio/high_mean": 0.0521149430423975, "clip_ratio/low_mean": 0.013554217293858528, "clip_ratio/low_min": 0.013554217293858528, "clip_ratio/region_mean": 0.06566916033625603, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 85.875, "completions/mean_terminated_length": 85.875, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.31395105831325054, "epoch": 0.11892999156524882, "frac_reward_zero_std": 0.0, "grad_norm": 9.514266967773438, "learning_rate": 1.0303030303030304e-06, "loss": -0.0069, "num_tokens": 6721217.0, "reward": 0.9182481169700623, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9182481169700623, "reward_meter_std": 0.11503338068723679, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.11503338813781738, "reward_total_composite_mean": 0.9182481169700623, "reward_total_composite_std": 0.11503338068723679, "reward_total_mean": 0.9182481169700623, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9182481169700623, "rewards/meter/std": 0.11503338068723679, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9182481169700623, "rewards/total_composite/std": 0.11503338068723679, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9983007907867432, "sampling/importance_sampling_ratio/min": 0.07875242084264755, "sampling/sampling_logp_difference/max": 2.5414462089538574, "sampling/sampling_logp_difference/mean": 0.06353027373552322, "step": 2961 }, { "clip_ratio/high_max": 0.001953125, "clip_ratio/high_mean": 0.001953125, "clip_ratio/low_mean": 0.0019841270986944437, "clip_ratio/low_min": 0.0019841270986944437, "clip_ratio/region_mean": 0.003937252098694444, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 63.625, "completions/mean_terminated_length": 63.625, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.009632107801735401, "epoch": 0.11897015704703377, "frac_reward_zero_std": 0.0, "grad_norm": 4.379561424255371, "learning_rate": 1.0272727272727274e-06, "loss": -0.0033, "num_tokens": 6722966.0, "reward": 0.9992475509643555, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992475509643555, "reward_meter_std": 0.00020958359527867287, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020958769891876727, "reward_total_composite_mean": 0.9992475509643555, "reward_total_composite_std": 0.00020958359527867287, "reward_total_mean": 0.9992475509643555, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992475509643555, "rewards/meter/std": 0.00020958359527867287, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992475509643555, "rewards/total_composite/std": 0.00020958359527867287, "sampling/importance_sampling_ratio/max": 1.7847542762756348, "sampling/importance_sampling_ratio/mean": 1.0014468431472778, "sampling/importance_sampling_ratio/min": 0.26296672224998474, "sampling/sampling_logp_difference/max": 1.3357278108596802, "sampling/sampling_logp_difference/mean": 0.00583175802603364, "step": 2962 }, { "clip_ratio/high_max": 0.023915290948934853, "clip_ratio/high_mean": 0.023915290948934853, "clip_ratio/low_mean": 0.004687500186264515, "clip_ratio/low_min": 0.004687500186264515, "clip_ratio/region_mean": 0.028602791135199368, "completions/clipped_ratio": 0.0, "completions/max_length": 160.0, "completions/max_terminated_length": 160.0, "completions/mean_length": 157.125, "completions/mean_terminated_length": 157.125, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "entropy": 0.3180831354111433, "epoch": 0.11901032252881873, "frac_reward_zero_std": 0.0, "grad_norm": 2.2698678970336914, "learning_rate": 1.0242424242424242e-06, "loss": 0.0074, "num_tokens": 6725671.0, "reward": 0.9811090230941772, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989545345306396, "reward_meter_std": 0.0008401040686294436, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.050323426723480225, "reward_total_composite_mean": 0.9811090230941772, "reward_total_composite_std": 0.05032341554760933, "reward_total_mean": 0.9811090230941772, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989545345306396, "rewards/meter/std": 0.0008401040686294436, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9811090230941772, "rewards/total_composite/std": 0.05032341554760933, "sampling/importance_sampling_ratio/max": 1.5869454145431519, "sampling/importance_sampling_ratio/mean": 1.0084627866744995, "sampling/importance_sampling_ratio/min": 0.19442002475261688, "sampling/sampling_logp_difference/max": 1.6377344131469727, "sampling/sampling_logp_difference/mean": 0.03372993692755699, "step": 2963 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/region_mean": 0.001953125, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.00610040313040372, "epoch": 0.11905048801060368, "frac_reward_zero_std": 0.0, "grad_norm": 1.2236340045928955, "learning_rate": 1.0212121212121213e-06, "loss": -0.0008, "num_tokens": 6727399.0, "reward": 0.9993772506713867, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993772506713867, "reward_meter_std": 6.269344157772139e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.270247104112059e-05, "reward_total_composite_mean": 0.9993772506713867, "reward_total_composite_std": 6.269344157772139e-05, "reward_total_mean": 0.9993772506713867, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993772506713867, "rewards/meter/std": 6.269344157772139e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993772506713867, "rewards/total_composite/std": 6.269344157772139e-05, "sampling/importance_sampling_ratio/max": 1.1811754703521729, "sampling/importance_sampling_ratio/mean": 0.9991680383682251, "sampling/importance_sampling_ratio/min": 0.5920082330703735, "sampling/sampling_logp_difference/max": 0.5242347717285156, "sampling/sampling_logp_difference/mean": 0.002508246572688222, "step": 2964 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.005542142200283706, "clip_ratio/low_min": 0.005542142200283706, "clip_ratio/region_mean": 0.007380377501249313, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.875, "completions/mean_terminated_length": 67.875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.0738078411668539, "epoch": 0.11909065349238863, "frac_reward_zero_std": 0.0, "grad_norm": 0.8599671721458435, "learning_rate": 1.0181818181818183e-06, "loss": -0.0025, "num_tokens": 6729286.0, "reward": 0.9994280338287354, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994280338287354, "reward_meter_std": 7.756530249025673e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.756453851470724e-05, "reward_total_composite_mean": 0.9994280338287354, "reward_total_composite_std": 7.756530249025673e-05, "reward_total_mean": 0.9994280338287354, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994280338287354, "rewards/meter/std": 7.756530249025673e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994280338287354, "rewards/total_composite/std": 7.756530249025673e-05, "sampling/importance_sampling_ratio/max": 1.371290683746338, "sampling/importance_sampling_ratio/mean": 0.9992300271987915, "sampling/importance_sampling_ratio/min": 0.4276924431324005, "sampling/sampling_logp_difference/max": 0.8493509292602539, "sampling/sampling_logp_difference/mean": 0.010112082585692406, "step": 2965 }, { "clip_ratio/high_max": 0.014326842967420816, "clip_ratio/high_mean": 0.014326842967420816, "clip_ratio/low_mean": 0.008422989631071687, "clip_ratio/low_min": 0.008422989631071687, "clip_ratio/region_mean": 0.022749832598492503, "completions/clipped_ratio": 0.0, "completions/max_length": 301.0, "completions/max_terminated_length": 301.0, "completions/mean_length": 296.375, "completions/mean_terminated_length": 296.375, "completions/min_length": 291.0, "completions/min_terminated_length": 291.0, "entropy": 0.3008219301700592, "epoch": 0.11913081897417359, "frac_reward_zero_std": 0.0, "grad_norm": 1.9423378705978394, "learning_rate": 1.0151515151515152e-06, "loss": 0.0035, "num_tokens": 6733705.0, "reward": 0.9196349382400513, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9927277565002441, "reward_meter_std": 0.018433313816785812, "reward_repeat_penalty_mean": 0.9264706373214722, "reward_repeat_penalty_std": 0.04159451648592949, "reward_std": 0.04263370484113693, "reward_total_composite_mean": 0.9196349382400513, "reward_total_composite_std": 0.042633701115846634, "reward_total_mean": 0.9196349382400513, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9927277565002441, "rewards/meter/std": 0.018433313816785812, "rewards/repeat_penalty/mean": 0.9264706373214722, "rewards/repeat_penalty/std": 0.04159451648592949, "rewards/total_composite/mean": 0.9196349382400513, "rewards/total_composite/std": 0.042633701115846634, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0065155029296875, "sampling/importance_sampling_ratio/min": 0.3544961214065552, "sampling/sampling_logp_difference/max": 1.037057876586914, "sampling/sampling_logp_difference/mean": 0.029081236571073532, "step": 2966 }, { "clip_ratio/high_max": 0.008167747058905661, "clip_ratio/high_mean": 0.008167747058905661, "clip_ratio/low_mean": 0.008021619636565447, "clip_ratio/low_min": 0.008021619636565447, "clip_ratio/region_mean": 0.016189366695471108, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 92.5, "completions/mean_terminated_length": 92.5, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.11041967011988163, "epoch": 0.11917098445595854, "frac_reward_zero_std": 0.0, "grad_norm": 2.049762487411499, "learning_rate": 1.0121212121212122e-06, "loss": 0.0063, "num_tokens": 6735749.0, "reward": 0.9977041482925415, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977041482925415, "reward_meter_std": 0.00018006080063059926, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001800469763111323, "reward_total_composite_mean": 0.9977041482925415, "reward_total_composite_std": 0.00018006080063059926, "reward_total_mean": 0.9977041482925415, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977041482925415, "rewards/meter/std": 0.00018006080063059926, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977041482925415, "rewards/total_composite/std": 0.00018006080063059926, "sampling/importance_sampling_ratio/max": 1.9789665937423706, "sampling/importance_sampling_ratio/mean": 1.0007275342941284, "sampling/importance_sampling_ratio/min": 0.34865638613700867, "sampling/sampling_logp_difference/max": 1.0536683797836304, "sampling/sampling_logp_difference/mean": 0.01588408276438713, "step": 2967 }, { "clip_ratio/high_max": 0.0354147027246654, "clip_ratio/high_mean": 0.0354147027246654, "clip_ratio/low_mean": 0.00747274118475616, "clip_ratio/low_min": 0.00747274118475616, "clip_ratio/region_mean": 0.04288744390942156, "completions/clipped_ratio": 0.0, "completions/max_length": 347.0, "completions/max_terminated_length": 347.0, "completions/mean_length": 337.5, "completions/mean_terminated_length": 337.5, "completions/min_length": 332.0, "completions/min_terminated_length": 332.0, "entropy": 0.4550737701356411, "epoch": 0.1192111499377435, "frac_reward_zero_std": 0.0, "grad_norm": 1.8433130979537964, "learning_rate": 1.0090909090909092e-06, "loss": 0.0007, "num_tokens": 6740073.0, "reward": 0.9847573637962341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978909492492676, "reward_meter_std": 0.0014596291584894061, "reward_repeat_penalty_mean": 0.9868420958518982, "reward_repeat_penalty_std": 0.024363677948713303, "reward_std": 0.024200553074479103, "reward_total_composite_mean": 0.9847573637962341, "reward_total_composite_std": 0.024200566112995148, "reward_total_mean": 0.9847573637962341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978909492492676, "rewards/meter/std": 0.0014596291584894061, "rewards/repeat_penalty/mean": 0.9868420958518982, "rewards/repeat_penalty/std": 0.024363677948713303, "rewards/total_composite/mean": 0.9847573637962341, "rewards/total_composite/std": 0.024200566112995148, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0109304189682007, "sampling/importance_sampling_ratio/min": 0.05878135934472084, "sampling/sampling_logp_difference/max": 2.833930492401123, "sampling/sampling_logp_difference/mean": 0.051781829446554184, "step": 2968 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.005584679980529472, "epoch": 0.11925131541952846, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.006060606060606e-06, "loss": 0.0, "num_tokens": 6741609.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0287812948226929, "sampling/importance_sampling_ratio/mean": 1.000121831893921, "sampling/importance_sampling_ratio/min": 0.9628793001174927, "sampling/sampling_logp_difference/max": 0.0378272607922554, "sampling/sampling_logp_difference/mean": 0.0006444337195716798, "step": 2969 }, { "clip_ratio/high_max": 0.006643245578743517, "clip_ratio/high_mean": 0.006643245578743517, "clip_ratio/low_mean": 0.005667578196153045, "clip_ratio/low_min": 0.005667578196153045, "clip_ratio/region_mean": 0.012310823774896562, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 132.0, "completions/mean_terminated_length": 132.0, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.09148383047431707, "epoch": 0.11929148090131342, "frac_reward_zero_std": 0.0, "grad_norm": 0.3558986186981201, "learning_rate": 1.0030303030303031e-06, "loss": 0.0005, "num_tokens": 6744177.0, "reward": 0.9994352459907532, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994352459907532, "reward_meter_std": 1.4545888916472904e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.4537939932779409e-05, "reward_total_composite_mean": 0.9994352459907532, "reward_total_composite_std": 1.4545888916472904e-05, "reward_total_mean": 0.9994352459907532, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994352459907532, "rewards/meter/std": 1.4545888916472904e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994352459907532, "rewards/total_composite/std": 1.4545888916472904e-05, "sampling/importance_sampling_ratio/max": 1.4119945764541626, "sampling/importance_sampling_ratio/mean": 1.0008372068405151, "sampling/importance_sampling_ratio/min": 0.5033237934112549, "sampling/sampling_logp_difference/max": 0.6865215301513672, "sampling/sampling_logp_difference/mean": 0.010489360429346561, "step": 2970 }, { "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/low_mean": 0.006377551006153226, "clip_ratio/low_min": 0.006377551006153226, "clip_ratio/region_mean": 0.007653061184100807, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 98.125, "completions/mean_terminated_length": 98.125, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.051710725761950016, "epoch": 0.11933164638309837, "frac_reward_zero_std": 0.0, "grad_norm": 0.8809167742729187, "learning_rate": 1.0000000000000002e-06, "loss": 0.001, "num_tokens": 6746402.0, "reward": 0.9994017481803894, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994017481803894, "reward_meter_std": 5.140481152920984e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.139390123076737e-05, "reward_total_composite_mean": 0.9994017481803894, "reward_total_composite_std": 5.140481152920984e-05, "reward_total_mean": 0.9994017481803894, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994017481803894, "rewards/meter/std": 5.140481152920984e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994017481803894, "rewards/total_composite/std": 5.140481152920984e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0018893480300903, "sampling/importance_sampling_ratio/min": 0.5932945609092712, "sampling/sampling_logp_difference/max": 1.0734617710113525, "sampling/sampling_logp_difference/mean": 0.011422020383179188, "step": 2971 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.010805643978528678, "epoch": 0.11937181186488333, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.96969696969697e-07, "loss": 0.0, "num_tokens": 6747826.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.056694507598877, "sampling/importance_sampling_ratio/mean": 1.000320315361023, "sampling/importance_sampling_ratio/min": 0.9198248982429504, "sampling/sampling_logp_difference/max": 0.08357194811105728, "sampling/sampling_logp_difference/mean": 0.0012666110415011644, "step": 2972 }, { "clip_ratio/high_max": 0.012083206791430712, "clip_ratio/high_mean": 0.012083206791430712, "clip_ratio/low_mean": 0.002494166372343898, "clip_ratio/low_min": 0.002494166372343898, "clip_ratio/region_mean": 0.01457737316377461, "completions/clipped_ratio": 0.0, "completions/max_length": 253.0, "completions/max_terminated_length": 253.0, "completions/mean_length": 248.875, "completions/mean_terminated_length": 248.875, "completions/min_length": 246.0, "completions/min_terminated_length": 246.0, "entropy": 0.2675408758223057, "epoch": 0.11941197734666828, "frac_reward_zero_std": 0.0, "grad_norm": 1.3921781778335571, "learning_rate": 9.93939393939394e-07, "loss": 0.0073, "num_tokens": 6751625.0, "reward": 0.9701830148696899, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990034699440002, "reward_meter_std": 0.00021533701510634273, "reward_repeat_penalty_mean": 0.9711538553237915, "reward_repeat_penalty_std": 0.039811473339796066, "reward_std": 0.039688270539045334, "reward_total_composite_mean": 0.9701830148696899, "reward_total_composite_std": 0.03968827798962593, "reward_total_mean": 0.9701830148696899, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990034699440002, "rewards/meter/std": 0.00021533701510634273, "rewards/repeat_penalty/mean": 0.9711538553237915, "rewards/repeat_penalty/std": 0.039811473339796066, "rewards/total_composite/mean": 0.9701830148696899, "rewards/total_composite/std": 0.03968827798962593, "sampling/importance_sampling_ratio/max": 1.81640625, "sampling/importance_sampling_ratio/mean": 1.0035536289215088, "sampling/importance_sampling_ratio/min": 0.29897233843803406, "sampling/sampling_logp_difference/max": 1.2074041366577148, "sampling/sampling_logp_difference/mean": 0.020545274019241333, "step": 2973 }, { "clip_ratio/high_max": 0.03571874415501952, "clip_ratio/high_mean": 0.03571874415501952, "clip_ratio/low_mean": 0.007359597599133849, "clip_ratio/low_min": 0.007359597599133849, "clip_ratio/region_mean": 0.04307834175415337, "completions/clipped_ratio": 0.0, "completions/max_length": 309.0, "completions/max_terminated_length": 309.0, "completions/mean_length": 292.75, "completions/mean_terminated_length": 292.75, "completions/min_length": 279.0, "completions/min_terminated_length": 279.0, "entropy": 0.4930482190102339, "epoch": 0.11945214282845323, "frac_reward_zero_std": 0.0, "grad_norm": 2.7557997703552246, "learning_rate": 9.90909090909091e-07, "loss": -0.0, "num_tokens": 6755543.0, "reward": 0.8287225961685181, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976239204406738, "reward_meter_std": 0.0013418762246146798, "reward_repeat_penalty_mean": 0.9558823108673096, "reward_repeat_penalty_std": 0.06852733343839645, "reward_std": 0.3412354588508606, "reward_total_composite_mean": 0.8287225961685181, "reward_total_composite_std": 0.3412354588508606, "reward_total_mean": 0.8287225961685181, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976239204406738, "rewards/meter/std": 0.0013418762246146798, "rewards/repeat_penalty/mean": 0.9558823108673096, "rewards/repeat_penalty/std": 0.06852733343839645, "rewards/total_composite/mean": 0.8287225961685181, "rewards/total_composite/std": 0.3412354588508606, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0052495002746582, "sampling/importance_sampling_ratio/min": 0.08845842629671097, "sampling/sampling_logp_difference/max": 2.425222635269165, "sampling/sampling_logp_difference/mean": 0.05460209771990776, "step": 2974 }, { "clip_ratio/high_max": 0.009191176504828036, "clip_ratio/high_mean": 0.009191176504828036, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.011029411805793643, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.07629173761233687, "epoch": 0.11949230831023819, "frac_reward_zero_std": 0.0, "grad_norm": 1.2692075967788696, "learning_rate": 9.87878787878788e-07, "loss": -0.0015, "num_tokens": 6757327.0, "reward": 0.9994783401489258, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994783401489258, "reward_meter_std": 9.867444896372035e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.867054177448153e-05, "reward_total_composite_mean": 0.9994783401489258, "reward_total_composite_std": 9.867444896372035e-05, "reward_total_mean": 0.9994783401489258, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994783401489258, "rewards/meter/std": 9.867444896372035e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994783401489258, "rewards/total_composite/std": 9.867444896372035e-05, "sampling/importance_sampling_ratio/max": 1.392788290977478, "sampling/importance_sampling_ratio/mean": 0.9991775751113892, "sampling/importance_sampling_ratio/min": 0.5110718607902527, "sampling/sampling_logp_difference/max": 0.6712450981140137, "sampling/sampling_logp_difference/mean": 0.009787042625248432, "step": 2975 }, { "clip_ratio/high_max": 0.006628788076341152, "clip_ratio/high_mean": 0.006628788076341152, "clip_ratio/low_mean": 0.000961538462433964, "clip_ratio/low_min": 0.000961538462433964, "clip_ratio/region_mean": 0.007590326538775116, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 131.75, "completions/mean_terminated_length": 131.75, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.07698143133893609, "epoch": 0.11953247379202314, "frac_reward_zero_std": 0.0, "grad_norm": 1.1907378435134888, "learning_rate": 9.84848484848485e-07, "loss": -0.0012, "num_tokens": 6759949.0, "reward": 0.9993938207626343, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993938207626343, "reward_meter_std": 7.404623465845361e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.40409450372681e-05, "reward_total_composite_mean": 0.9993938207626343, "reward_total_composite_std": 7.404623465845361e-05, "reward_total_mean": 0.9993938207626343, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993938207626343, "rewards/meter/std": 7.404623465845361e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993938207626343, "rewards/total_composite/std": 7.404623465845361e-05, "sampling/importance_sampling_ratio/max": 1.4190884828567505, "sampling/importance_sampling_ratio/mean": 1.0010247230529785, "sampling/importance_sampling_ratio/min": 0.5627366304397583, "sampling/sampling_logp_difference/max": 0.5749435424804688, "sampling/sampling_logp_difference/mean": 0.008480089716613293, "step": 2976 }, { "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/region_mean": 0.0038265305338427424, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.0308038501534611, "epoch": 0.1195726392738081, "frac_reward_zero_std": 0.0, "grad_norm": 0.025195496156811714, "learning_rate": 9.818181818181818e-07, "loss": 0.0003, "num_tokens": 6762117.0, "reward": 0.9994038343429565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994038343429565, "reward_meter_std": 3.7851127672183793e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.7985312246746616e-06, "reward_total_composite_mean": 0.9994038343429565, "reward_total_composite_std": 3.7851127672183793e-06, "reward_total_mean": 0.9994038343429565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994038343429565, "rewards/meter/std": 3.7851127672183793e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994038343429565, "rewards/total_composite/std": 3.7851127672183793e-06, "sampling/importance_sampling_ratio/max": 1.2943319082260132, "sampling/importance_sampling_ratio/mean": 1.000174880027771, "sampling/importance_sampling_ratio/min": 0.8075622916221619, "sampling/sampling_logp_difference/max": 0.2579946517944336, "sampling/sampling_logp_difference/mean": 0.0029753749258816242, "step": 2977 }, { "clip_ratio/high_max": 0.02096639317460358, "clip_ratio/high_mean": 0.02096639317460358, "clip_ratio/low_mean": 0.003756533144041896, "clip_ratio/low_min": 0.003756533144041896, "clip_ratio/region_mean": 0.024722926318645477, "completions/clipped_ratio": 0.0, "completions/max_length": 173.0, "completions/max_terminated_length": 173.0, "completions/mean_length": 167.125, "completions/mean_terminated_length": 167.125, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.30791075527668, "epoch": 0.11961280475559305, "frac_reward_zero_std": 0.0, "grad_norm": 3.375030994415283, "learning_rate": 9.787878787878788e-07, "loss": -0.0016, "num_tokens": 6764878.0, "reward": 0.9690313339233398, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9967660903930664, "reward_meter_std": 0.00537865748628974, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.050591256469488144, "reward_total_composite_mean": 0.9690313339233398, "reward_total_composite_std": 0.05059126019477844, "reward_total_mean": 0.9690313339233398, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9967660903930664, "rewards/meter/std": 0.00537865748628974, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9690313339233398, "rewards/total_composite/std": 0.05059126019477844, "sampling/importance_sampling_ratio/max": 1.8077061176300049, "sampling/importance_sampling_ratio/mean": 1.0075362920761108, "sampling/importance_sampling_ratio/min": 0.2535859942436218, "sampling/sampling_logp_difference/max": 1.3720521926879883, "sampling/sampling_logp_difference/mean": 0.03752928227186203, "step": 2978 }, { "clip_ratio/high_max": 0.02296627010218799, "clip_ratio/high_mean": 0.02296627010218799, "clip_ratio/low_mean": 0.008778364514000714, "clip_ratio/low_min": 0.008778364514000714, "clip_ratio/region_mean": 0.031744634616188705, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 70.875, "completions/mean_terminated_length": 70.875, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.27640925347805023, "epoch": 0.119652970237378, "frac_reward_zero_std": 0.0, "grad_norm": 6.203634738922119, "learning_rate": 9.757575757575759e-07, "loss": 0.015, "num_tokens": 6766789.0, "reward": 0.9866373538970947, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9866373538970947, "reward_meter_std": 0.01733180694282055, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.017331810668110847, "reward_total_composite_mean": 0.9866373538970947, "reward_total_composite_std": 0.01733180694282055, "reward_total_mean": 0.9866373538970947, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9866373538970947, "rewards/meter/std": 0.01733180694282055, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9866373538970947, "rewards/total_composite/std": 0.01733180694282055, "sampling/importance_sampling_ratio/max": 1.5757840871810913, "sampling/importance_sampling_ratio/mean": 1.0079153776168823, "sampling/importance_sampling_ratio/min": 0.5158604979515076, "sampling/sampling_logp_difference/max": 0.6619188785552979, "sampling/sampling_logp_difference/mean": 0.025828473269939423, "step": 2979 }, { "clip_ratio/high_max": 0.007247899193316698, "clip_ratio/high_mean": 0.007247899193316698, "clip_ratio/low_mean": 0.0053571430034935474, "clip_ratio/low_min": 0.0053571430034935474, "clip_ratio/region_mean": 0.012605042196810246, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.5, "completions/mean_terminated_length": 68.5, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.11373988864943385, "epoch": 0.11969313571916296, "frac_reward_zero_std": 0.0, "grad_norm": 5.659327507019043, "learning_rate": 9.72727272727273e-07, "loss": 0.0156, "num_tokens": 6768713.0, "reward": 0.9975705742835999, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975705742835999, "reward_meter_std": 0.004985023755580187, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004985013976693153, "reward_total_composite_mean": 0.9975705742835999, "reward_total_composite_std": 0.004985023755580187, "reward_total_mean": 0.9975705742835999, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975705742835999, "rewards/meter/std": 0.004985023755580187, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975705742835999, "rewards/total_composite/std": 0.004985023755580187, "sampling/importance_sampling_ratio/max": 1.6995705366134644, "sampling/importance_sampling_ratio/mean": 1.0029165744781494, "sampling/importance_sampling_ratio/min": 0.3158693015575409, "sampling/sampling_logp_difference/max": 1.1524267196655273, "sampling/sampling_logp_difference/mean": 0.016337888315320015, "step": 2980 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.0012292767642065883, "epoch": 0.11973330120094791, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.696969696969698e-07, "loss": 0.0, "num_tokens": 6770249.0, "reward": 0.9943599104881287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9943599104881287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0023645162582397, "sampling/importance_sampling_ratio/mean": 1.0001403093338013, "sampling/importance_sampling_ratio/min": 0.9999930262565613, "sampling/sampling_logp_difference/max": 0.0023617574479430914, "sampling/sampling_logp_difference/mean": 0.0001402015914209187, "step": 2981 }, { "clip_ratio/high_max": 0.02634434588253498, "clip_ratio/high_mean": 0.02634434588253498, "clip_ratio/low_mean": 0.0032051282469183207, "clip_ratio/low_min": 0.0032051282469183207, "clip_ratio/region_mean": 0.0295494741294533, "completions/clipped_ratio": 0.0, "completions/max_length": 121.0, "completions/max_terminated_length": 121.0, "completions/mean_length": 118.25, "completions/mean_terminated_length": 118.25, "completions/min_length": 117.0, "completions/min_terminated_length": 117.0, "entropy": 0.2766532879322767, "epoch": 0.11977346668273287, "frac_reward_zero_std": 0.0, "grad_norm": 2.0111052989959717, "learning_rate": 9.666666666666668e-07, "loss": -0.0015, "num_tokens": 6772571.0, "reward": 0.9989663362503052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989663362503052, "reward_meter_std": 0.0008744557853788137, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008744720835238695, "reward_total_composite_mean": 0.9989663362503052, "reward_total_composite_std": 0.0008744557853788137, "reward_total_mean": 0.9989663362503052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989663362503052, "rewards/meter/std": 0.0008744557853788137, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989663362503052, "rewards/total_composite/std": 0.0008744557853788137, "sampling/importance_sampling_ratio/max": 1.6112860441207886, "sampling/importance_sampling_ratio/mean": 1.0011169910430908, "sampling/importance_sampling_ratio/min": 0.18890489637851715, "sampling/sampling_logp_difference/max": 1.6665115356445312, "sampling/sampling_logp_difference/mean": 0.03213733434677124, "step": 2982 }, { "clip_ratio/high_max": 0.016288378508761525, "clip_ratio/high_mean": 0.016288378508761525, "clip_ratio/low_mean": 0.0061881187139078975, "clip_ratio/low_min": 0.0061881187139078975, "clip_ratio/region_mean": 0.022476497222669423, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 100.5, "completions/mean_terminated_length": 100.5, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.17713584937155247, "epoch": 0.11981363216451782, "frac_reward_zero_std": 0.0, "grad_norm": 4.000948905944824, "learning_rate": 9.636363636363636e-07, "loss": 0.0056, "num_tokens": 6774679.0, "reward": 0.9243372678756714, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993005990982056, "reward_meter_std": 0.00027800656971521676, "reward_repeat_penalty_mean": 0.9249999523162842, "reward_repeat_penalty_std": 0.1035098284482956, "reward_std": 0.10328476876020432, "reward_total_composite_mean": 0.9243372678756714, "reward_total_composite_std": 0.1032847911119461, "reward_total_mean": 0.9243372678756714, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993005990982056, "rewards/meter/std": 0.00027800656971521676, "rewards/repeat_penalty/mean": 0.9249999523162842, "rewards/repeat_penalty/std": 0.1035098284482956, "rewards/total_composite/mean": 0.9243372678756714, "rewards/total_composite/std": 0.1032847911119461, "sampling/importance_sampling_ratio/max": 1.835827350616455, "sampling/importance_sampling_ratio/mean": 1.0000145435333252, "sampling/importance_sampling_ratio/min": 0.3460240960121155, "sampling/sampling_logp_difference/max": 1.0612468719482422, "sampling/sampling_logp_difference/mean": 0.023382917046546936, "step": 2983 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.003427958843531087, "epoch": 0.11985379764630277, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.606060606060607e-07, "loss": 0.0, "num_tokens": 6776207.0, "reward": 0.9980231523513794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9980231523513794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.004080891609192, "sampling/importance_sampling_ratio/mean": 1.0001977682113647, "sampling/importance_sampling_ratio/min": 0.995279848575592, "sampling/sampling_logp_difference/max": 0.00473133847117424, "sampling/sampling_logp_difference/mean": 0.0002929195179603994, "step": 2984 }, { "clip_ratio/high_max": 0.011204644571989775, "clip_ratio/high_mean": 0.011204644571989775, "clip_ratio/low_mean": 0.018017811933532357, "clip_ratio/low_min": 0.018017811933532357, "clip_ratio/region_mean": 0.029222456505522132, "completions/clipped_ratio": 0.0, "completions/max_length": 171.0, "completions/max_terminated_length": 171.0, "completions/mean_length": 167.375, "completions/mean_terminated_length": 167.375, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.2825808133929968, "epoch": 0.11989396312808773, "frac_reward_zero_std": 0.0, "grad_norm": 2.747091770172119, "learning_rate": 9.575757575757577e-07, "loss": 0.0071, "num_tokens": 6779042.0, "reward": 0.9275848865509033, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9969633221626282, "reward_meter_std": 0.003777019679546356, "reward_repeat_penalty_mean": 0.9305555820465088, "reward_repeat_penalty_std": 0.05750546231865883, "reward_std": 0.05462638661265373, "reward_total_composite_mean": 0.9275848865509033, "reward_total_composite_std": 0.05462638661265373, "reward_total_mean": 0.9275848865509033, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9969633221626282, "rewards/meter/std": 0.003777019679546356, "rewards/repeat_penalty/mean": 0.9305555820465088, "rewards/repeat_penalty/std": 0.05750546231865883, "rewards/total_composite/mean": 0.9275848865509033, "rewards/total_composite/std": 0.05462638661265373, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004103422164917, "sampling/importance_sampling_ratio/min": 0.20248271524906158, "sampling/sampling_logp_difference/max": 1.5971007347106934, "sampling/sampling_logp_difference/mean": 0.03595529869198799, "step": 2985 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.00046791824570391327, "epoch": 0.11993412860987268, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.545454545454548e-07, "loss": 0.0, "num_tokens": 6780498.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0009034872055054, "sampling/importance_sampling_ratio/mean": 1.0000447034835815, "sampling/importance_sampling_ratio/min": 0.9995164275169373, "sampling/sampling_logp_difference/max": 0.0009029797511175275, "sampling/sampling_logp_difference/mean": 5.2063212933717296e-05, "step": 2986 }, { "clip_ratio/high_max": 0.0034966744715347886, "clip_ratio/high_mean": 0.0034966744715347886, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0034966744715347886, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.125, "completions/mean_terminated_length": 71.125, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.08504062425345182, "epoch": 0.11997429409165764, "frac_reward_zero_std": 0.0, "grad_norm": 2.3871347904205322, "learning_rate": 9.515151515151516e-07, "loss": -0.004, "num_tokens": 6782659.0, "reward": 0.9993984699249268, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993984699249268, "reward_meter_std": 9.4733273726888e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.472780220676214e-05, "reward_total_composite_mean": 0.9993984699249268, "reward_total_composite_std": 9.4733273726888e-05, "reward_total_mean": 0.9993984699249268, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993984699249268, "rewards/meter/std": 9.4733273726888e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993984699249268, "rewards/total_composite/std": 9.4733273726888e-05, "sampling/importance_sampling_ratio/max": 1.32568359375, "sampling/importance_sampling_ratio/mean": 1.0039633512496948, "sampling/importance_sampling_ratio/min": 0.5364948511123657, "sampling/sampling_logp_difference/max": 0.6226983070373535, "sampling/sampling_logp_difference/mean": 0.010385324247181416, "step": 2987 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0022001640463713557, "epoch": 0.12001445957344259, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.484848484848485e-07, "loss": 0.0, "num_tokens": 6784427.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0033867359161377, "sampling/importance_sampling_ratio/mean": 1.0002238750457764, "sampling/importance_sampling_ratio/min": 0.9995951652526855, "sampling/sampling_logp_difference/max": 0.003381013870239258, "sampling/sampling_logp_difference/mean": 0.00022603126126341522, "step": 2988 }, { "clip_ratio/high_max": 0.02337344060651958, "clip_ratio/high_mean": 0.02337344060651958, "clip_ratio/low_mean": 0.0028013800038024783, "clip_ratio/low_min": 0.0028013800038024783, "clip_ratio/region_mean": 0.026174820610322058, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 133.375, "completions/mean_terminated_length": 133.375, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.24928219243884087, "epoch": 0.12005462505522754, "frac_reward_zero_std": 0.0, "grad_norm": 3.191429615020752, "learning_rate": 9.454545454545455e-07, "loss": 0.0025, "num_tokens": 6786878.0, "reward": 0.9622775316238403, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979535937309265, "reward_meter_std": 0.0034614717587828636, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.06554586440324783, "reward_total_composite_mean": 0.9622775316238403, "reward_total_composite_std": 0.06554586440324783, "reward_total_mean": 0.9622775316238403, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979535937309265, "rewards/meter/std": 0.0034614717587828636, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9622775316238403, "rewards/total_composite/std": 0.06554586440324783, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9995792508125305, "sampling/importance_sampling_ratio/min": 0.08252352476119995, "sampling/sampling_logp_difference/max": 2.4946718215942383, "sampling/sampling_logp_difference/mean": 0.03642260655760765, "step": 2989 }, { "clip_ratio/high_max": 0.008035918464884162, "clip_ratio/high_mean": 0.008035918464884162, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/region_mean": 0.012068176409229636, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 92.875, "completions/mean_terminated_length": 92.875, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.09223943296819925, "epoch": 0.1200947905370125, "frac_reward_zero_std": 0.0, "grad_norm": 3.7938873767852783, "learning_rate": 9.424242424242425e-07, "loss": -0.0015, "num_tokens": 6789021.0, "reward": 0.997610330581665, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997610330581665, "reward_meter_std": 0.00025703865685500205, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002570512588135898, "reward_total_composite_mean": 0.997610330581665, "reward_total_composite_std": 0.00025703865685500205, "reward_total_mean": 0.997610330581665, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997610330581665, "rewards/meter/std": 0.00025703865685500205, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997610330581665, "rewards/total_composite/std": 0.00025703865685500205, "sampling/importance_sampling_ratio/max": 1.416971206665039, "sampling/importance_sampling_ratio/mean": 0.9976533055305481, "sampling/importance_sampling_ratio/min": 0.33530673384666443, "sampling/sampling_logp_difference/max": 1.0927095413208008, "sampling/sampling_logp_difference/mean": 0.01364238653331995, "step": 2990 }, { "clip_ratio/high_max": 0.0056535504991188645, "clip_ratio/high_mean": 0.0056535504991188645, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.007547489949502051, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.14550988655537367, "epoch": 0.12013495601879745, "frac_reward_zero_std": 0.0, "grad_norm": 2.051816463470459, "learning_rate": 9.393939393939395e-07, "loss": -0.0063, "num_tokens": 6790862.0, "reward": 0.9928753972053528, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9928753972053528, "reward_meter_std": 0.0007522930391132832, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000752286403439939, "reward_total_composite_mean": 0.9928753972053528, "reward_total_composite_std": 0.0007522930391132832, "reward_total_mean": 0.9928753972053528, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9928753972053528, "rewards/meter/std": 0.0007522930391132832, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9928753972053528, "rewards/total_composite/std": 0.0007522930391132832, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0033050775527954, "sampling/importance_sampling_ratio/min": 0.16668224334716797, "sampling/sampling_logp_difference/max": 1.791666030883789, "sampling/sampling_logp_difference/mean": 0.024633267894387245, "step": 2991 }, { "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006147540640085936, "completions/clipped_ratio": 0.0, "completions/max_length": 62.0, "completions/max_terminated_length": 62.0, "completions/mean_length": 61.125, "completions/mean_terminated_length": 61.125, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.025716731324791908, "epoch": 0.1201751215005824, "frac_reward_zero_std": 0.0, "grad_norm": 3.857464075088501, "learning_rate": 9.363636363636365e-07, "loss": 0.0024, "num_tokens": 6792639.0, "reward": 0.9971661567687988, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9971661567687988, "reward_meter_std": 0.00047978354268707335, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004797954170498997, "reward_total_composite_mean": 0.9971661567687988, "reward_total_composite_std": 0.00047978354268707335, "reward_total_mean": 0.9971661567687988, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9971661567687988, "rewards/meter/std": 0.00047978354268707335, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9971661567687988, "rewards/total_composite/std": 0.00047978354268707335, "sampling/importance_sampling_ratio/max": 1.1078962087631226, "sampling/importance_sampling_ratio/mean": 0.9966216087341309, "sampling/importance_sampling_ratio/min": 0.18380838632583618, "sampling/sampling_logp_difference/max": 1.693861484527588, "sampling/sampling_logp_difference/mean": 0.00791865587234497, "step": 2992 }, { "clip_ratio/high_max": 0.014776158845052123, "clip_ratio/high_mean": 0.014776158845052123, "clip_ratio/low_mean": 0.011799397761933506, "clip_ratio/low_min": 0.011799397761933506, "clip_ratio/region_mean": 0.02657555660698563, "completions/clipped_ratio": 0.0, "completions/max_length": 332.0, "completions/max_terminated_length": 332.0, "completions/mean_length": 328.625, "completions/mean_terminated_length": 328.625, "completions/min_length": 325.0, "completions/min_terminated_length": 325.0, "entropy": 0.3455524481832981, "epoch": 0.12021528698236736, "frac_reward_zero_std": 0.0, "grad_norm": 1.920817255973816, "learning_rate": 9.333333333333334e-07, "loss": 0.0031, "num_tokens": 6797268.0, "reward": 0.9133861064910889, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988096952438354, "reward_meter_std": 0.0004608924500644207, "reward_repeat_penalty_mean": 0.9144736528396606, "reward_repeat_penalty_std": 0.05582422763109207, "reward_std": 0.055794164538383484, "reward_total_composite_mean": 0.9133861064910889, "reward_total_composite_std": 0.05579417571425438, "reward_total_mean": 0.9133861064910889, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988096952438354, "rewards/meter/std": 0.0004608924500644207, "rewards/repeat_penalty/mean": 0.9144736528396606, "rewards/repeat_penalty/std": 0.05582422763109207, "rewards/total_composite/mean": 0.9133861064910889, "rewards/total_composite/std": 0.05579417571425438, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007267951965332, "sampling/importance_sampling_ratio/min": 0.2541518211364746, "sampling/sampling_logp_difference/max": 1.3698234558105469, "sampling/sampling_logp_difference/mean": 0.03226817771792412, "step": 2993 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0016921081114560366, "epoch": 0.12025545246415231, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.303030303030304e-07, "loss": 0.0, "num_tokens": 6798980.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0038442611694336, "sampling/importance_sampling_ratio/mean": 1.0002259016036987, "sampling/importance_sampling_ratio/min": 0.99946129322052, "sampling/sampling_logp_difference/max": 0.0038368557579815388, "sampling/sampling_logp_difference/mean": 0.00023179415438789874, "step": 2994 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.00262832005682867, "epoch": 0.12029561794593727, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.272727272727273e-07, "loss": 0.0, "num_tokens": 6800420.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0056308507919312, "sampling/importance_sampling_ratio/mean": 1.0001194477081299, "sampling/importance_sampling_ratio/min": 0.9919206500053406, "sampling/sampling_logp_difference/max": 0.008112169802188873, "sampling/sampling_logp_difference/mean": 0.00022380216978490353, "step": 2995 }, { "clip_ratio/high_max": 0.0009469697251915932, "clip_ratio/high_mean": 0.0009469697251915932, "clip_ratio/low_mean": 0.000961538462433964, "clip_ratio/low_min": 0.000961538462433964, "clip_ratio/region_mean": 0.0019085081876255572, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 131.875, "completions/mean_terminated_length": 131.875, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.0709062721580267, "epoch": 0.12033578342772222, "frac_reward_zero_std": 0.0, "grad_norm": 0.4544410705566406, "learning_rate": 9.242424242424244e-07, "loss": -0.0, "num_tokens": 6802843.0, "reward": 0.9994057416915894, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994057416915894, "reward_meter_std": 4.619282117346302e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.6202192606870085e-05, "reward_total_composite_mean": 0.9994057416915894, "reward_total_composite_std": 4.619282117346302e-05, "reward_total_mean": 0.9994057416915894, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994057416915894, "rewards/meter/std": 4.619282117346302e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994057416915894, "rewards/total_composite/std": 4.619282117346302e-05, "sampling/importance_sampling_ratio/max": 1.366064429283142, "sampling/importance_sampling_ratio/mean": 1.0020174980163574, "sampling/importance_sampling_ratio/min": 0.3591609001159668, "sampling/sampling_logp_difference/max": 1.0239849090576172, "sampling/sampling_logp_difference/mean": 0.007839047349989414, "step": 2996 }, { "clip_ratio/high_max": 0.012055294588208199, "clip_ratio/high_mean": 0.012055294588208199, "clip_ratio/low_mean": 0.010527235455811024, "clip_ratio/low_min": 0.010527235455811024, "clip_ratio/region_mean": 0.022582530044019222, "completions/clipped_ratio": 0.5, "completions/max_length": 512.0, "completions/max_terminated_length": 512.0, "completions/mean_length": 507.5, "completions/mean_terminated_length": 503.0, "completions/min_length": 497.0, "completions/min_terminated_length": 497.0, "entropy": 0.33224867284297943, "epoch": 0.12037594890950717, "frac_reward_zero_std": 0.0, "grad_norm": 2.213428020477295, "learning_rate": 9.212121212121213e-07, "loss": -0.0832, "num_tokens": 6806751.0, "reward": 0.8312985897064209, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8472222089767456, "reward_count_adherence_std": 0.0257172379642725, "reward_meter_mean": 0.9938823580741882, "reward_meter_std": 0.00720559898763895, "reward_repeat_penalty_mean": 0.9873470664024353, "reward_repeat_penalty_std": 0.01747622899711132, "reward_std": 0.026868147775530815, "reward_total_composite_mean": 0.8312985897064209, "reward_total_composite_std": 0.026868145912885666, "reward_total_mean": 0.8312985897064209, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8472222089767456, "rewards/count_adherence/std": 0.0257172379642725, "rewards/meter/mean": 0.9938823580741882, "rewards/meter/std": 0.00720559898763895, "rewards/repeat_penalty/mean": 0.9873470664024353, "rewards/repeat_penalty/std": 0.01747622899711132, "rewards/total_composite/mean": 0.8312985897064209, "rewards/total_composite/std": 0.026868145912885666, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0148636102676392, "sampling/importance_sampling_ratio/min": 0.10614575445652008, "sampling/sampling_logp_difference/max": 2.2429420948028564, "sampling/sampling_logp_difference/mean": 0.06921987235546112, "step": 2997 }, { "clip_ratio/high_max": 0.03046388761140406, "clip_ratio/high_mean": 0.03046388761140406, "clip_ratio/low_mean": 0.010312846396118402, "clip_ratio/low_min": 0.010312846396118402, "clip_ratio/region_mean": 0.040776734007522464, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 86.0, "completions/mean_terminated_length": 86.0, "completions/min_length": 84.0, "completions/min_terminated_length": 84.0, "entropy": 0.23842502385377884, "epoch": 0.12041611439129213, "frac_reward_zero_std": 0.0, "grad_norm": 5.6068010330200195, "learning_rate": 9.181818181818182e-07, "loss": -0.0063, "num_tokens": 6808791.0, "reward": 0.9505056738853455, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9505056738853455, "reward_meter_std": 0.01044482085853815, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.010444814339280128, "reward_total_composite_mean": 0.9505056738853455, "reward_total_composite_std": 0.01044482085853815, "reward_total_mean": 0.9505056738853455, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9505056738853455, "rewards/meter/std": 0.01044482085853815, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9505056738853455, "rewards/total_composite/std": 0.01044482085853815, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.004783272743225, "sampling/importance_sampling_ratio/min": 0.1372566968202591, "sampling/sampling_logp_difference/max": 1.9859023094177246, "sampling/sampling_logp_difference/mean": 0.046680573374032974, "step": 2998 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0032808549876790494, "epoch": 0.12045627987307708, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 9.151515151515153e-07, "loss": 0.0, "num_tokens": 6810823.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0093164443969727, "sampling/importance_sampling_ratio/mean": 1.0003594160079956, "sampling/importance_sampling_ratio/min": 0.9973099231719971, "sampling/sampling_logp_difference/max": 0.009273192845284939, "sampling/sampling_logp_difference/mean": 0.0003676189517136663, "step": 2999 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.002314814832061529, "clip_ratio/low_min": 0.002314814832061529, "clip_ratio/region_mean": 0.002314814832061529, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.01205012173159048, "epoch": 0.12049644535486204, "frac_reward_zero_std": 0.0, "grad_norm": 2.610877513885498, "learning_rate": 9.121212121212122e-07, "loss": -0.003, "num_tokens": 6812551.0, "reward": 0.7007421255111694, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7007421255111694, "reward_meter_std": 0.24578078091144562, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2457807958126068, "reward_total_composite_mean": 0.7007421255111694, "reward_total_composite_std": 0.24578078091144562, "reward_total_mean": 0.7007421255111694, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7007421255111694, "rewards/meter/std": 0.24578078091144562, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7007421255111694, "rewards/total_composite/std": 0.24578078091144562, "sampling/importance_sampling_ratio/max": 1.2230128049850464, "sampling/importance_sampling_ratio/mean": 0.9997680187225342, "sampling/importance_sampling_ratio/min": 0.22648018598556519, "sampling/sampling_logp_difference/max": 1.485097885131836, "sampling/sampling_logp_difference/mean": 0.0048951394855976105, "step": 3000 }, { "epoch": 0.12049644535486204, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/max_length": 421.53846153846155, "eval_completions/max_terminated_length": 412.7692307692308, "eval_completions/mean_length": 213.93269230769232, "eval_completions/mean_terminated_length": 210.74725341796875, "eval_completions/min_length": 61.23076923076923, "eval_completions/min_terminated_length": 61.23076923076923, "eval_entropy": 0.3522038482702695, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 6812551.0, "eval_reward": 0.7239083968676053, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9582245624982394, "eval_reward_count_adherence_std": 0.06270856238328494, "eval_reward_meter_mean": 0.7926333088141221, "eval_reward_meter_std": 0.3242100785629681, "eval_reward_repeat_penalty_mean": 0.9439093745671786, "eval_reward_repeat_penalty_std": 0.07702494555940995, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7239083968676053, "eval_reward_total_composite_std": 0.3284210069821431, "eval_reward_total_mean": 0.7239083968676053, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9582245624982394, "eval_rewards/count_adherence/std": 0.06270856238328494, "eval_rewards/meter/mean": 0.7926333088141221, "eval_rewards/meter/std": 0.3242100785629681, "eval_rewards/repeat_penalty/mean": 0.9439093745671786, "eval_rewards/repeat_penalty/std": 0.07702494555940995, "eval_rewards/total_composite/mean": 0.7239083968676053, "eval_rewards/total_composite/std": 0.3284210069821431, "eval_runtime": 79.4116, "eval_samples_per_second": 1.31, "eval_sampling/importance_sampling_ratio/max": 1.577669803912823, "eval_sampling/importance_sampling_ratio/mean": 1.0090940640522883, "eval_sampling/importance_sampling_ratio/min": 0.3013949474463096, "eval_sampling/sampling_logp_difference/max": 1.215673538354727, "eval_sampling/sampling_logp_difference/mean": 0.03129683048106157, "eval_steps_per_second": 0.164, "step": 3000 }, { "clip_ratio/high_max": 0.03723430214449763, "clip_ratio/high_mean": 0.03723430214449763, "clip_ratio/low_mean": 0.014480874873697758, "clip_ratio/low_min": 0.014480874873697758, "clip_ratio/region_mean": 0.05171517701819539, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 62.875, "completions/mean_terminated_length": 62.875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.28370110131800175, "epoch": 0.12053661083664699, "frac_reward_zero_std": 0.0, "grad_norm": 12.373333930969238, "learning_rate": 9.090909090909091e-07, "loss": -0.0098, "num_tokens": 6814326.0, "reward": 0.8256336450576782, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8256336450576782, "reward_meter_std": 0.3046148419380188, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.3046148419380188, "reward_total_composite_mean": 0.8256336450576782, "reward_total_composite_std": 0.3046148419380188, "reward_total_mean": 0.8256336450576782, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8256336450576782, "rewards/meter/std": 0.3046148419380188, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8256336450576782, "rewards/total_composite/std": 0.3046148419380188, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0055382251739502, "sampling/importance_sampling_ratio/min": 0.16510051488876343, "sampling/sampling_logp_difference/max": 1.8012008666992188, "sampling/sampling_logp_difference/mean": 0.06896661221981049, "step": 3001 }, { "clip_ratio/high_max": 0.01416189968585968, "clip_ratio/high_mean": 0.01416189968585968, "clip_ratio/low_mean": 0.0031650641467422247, "clip_ratio/low_min": 0.0031650641467422247, "clip_ratio/region_mean": 0.017326963832601905, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 78.625, "completions/mean_terminated_length": 78.625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.18326533026993275, "epoch": 0.12057677631843194, "frac_reward_zero_std": 0.0, "grad_norm": 3.2883293628692627, "learning_rate": 9.060606060606062e-07, "loss": 0.0072, "num_tokens": 6816267.0, "reward": 0.9988808631896973, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988808631896973, "reward_meter_std": 0.00034191427403129637, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00034191476879641414, "reward_total_composite_mean": 0.9988808631896973, "reward_total_composite_std": 0.00034191427403129637, "reward_total_mean": 0.9988808631896973, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988808631896973, "rewards/meter/std": 0.00034191427403129637, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988808631896973, "rewards/total_composite/std": 0.00034191427403129637, "sampling/importance_sampling_ratio/max": 1.6066337823867798, "sampling/importance_sampling_ratio/mean": 1.0053126811981201, "sampling/importance_sampling_ratio/min": 0.3121684789657593, "sampling/sampling_logp_difference/max": 1.1642122268676758, "sampling/sampling_logp_difference/mean": 0.02113337628543377, "step": 3002 }, { "clip_ratio/high_max": 0.012675640406087041, "clip_ratio/high_mean": 0.012675640406087041, "clip_ratio/low_mean": 0.008140756515786052, "clip_ratio/low_min": 0.008140756515786052, "clip_ratio/region_mean": 0.020816396921873093, "completions/clipped_ratio": 0.0, "completions/max_length": 142.0, "completions/max_terminated_length": 142.0, "completions/mean_length": 137.125, "completions/mean_terminated_length": 137.125, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.3485318683087826, "epoch": 0.1206169418002169, "frac_reward_zero_std": 0.0, "grad_norm": 3.6394121646881104, "learning_rate": 9.030303030303031e-07, "loss": 0.0131, "num_tokens": 6818772.0, "reward": 0.8669579029083252, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9025046229362488, "reward_meter_std": 0.2555387020111084, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.24918241798877716, "reward_total_composite_mean": 0.8669579029083252, "reward_total_composite_std": 0.24918241798877716, "reward_total_mean": 0.8669579029083252, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9025046229362488, "rewards/meter/std": 0.2555387020111084, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.8669579029083252, "rewards/total_composite/std": 0.24918241798877716, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0107687711715698, "sampling/importance_sampling_ratio/min": 0.3383774161338806, "sampling/sampling_logp_difference/max": 1.0835933685302734, "sampling/sampling_logp_difference/mean": 0.03274257108569145, "step": 3003 }, { "clip_ratio/high_max": 0.00777464872226119, "clip_ratio/high_mean": 0.00777464872226119, "clip_ratio/low_mean": 0.002914663462433964, "clip_ratio/low_min": 0.002914663462433964, "clip_ratio/region_mean": 0.010689312184695154, "completions/clipped_ratio": 0.0, "completions/max_length": 130.0, "completions/max_terminated_length": 130.0, "completions/mean_length": 128.75, "completions/mean_terminated_length": 128.75, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.1789141520857811, "epoch": 0.12065710728200185, "frac_reward_zero_std": 0.0, "grad_norm": 4.322078704833984, "learning_rate": 9.000000000000001e-07, "loss": 0.0001, "num_tokens": 6821058.0, "reward": 0.9970972537994385, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9970972537994385, "reward_meter_std": 0.001955085899680853, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0019550775177776814, "reward_total_composite_mean": 0.9970972537994385, "reward_total_composite_std": 0.001955085899680853, "reward_total_mean": 0.9970972537994385, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9970972537994385, "rewards/meter/std": 0.001955085899680853, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9970972537994385, "rewards/total_composite/std": 0.001955085899680853, "sampling/importance_sampling_ratio/max": 1.4015967845916748, "sampling/importance_sampling_ratio/mean": 1.0051413774490356, "sampling/importance_sampling_ratio/min": 0.3481079638004303, "sampling/sampling_logp_difference/max": 1.0552425384521484, "sampling/sampling_logp_difference/mean": 0.021072760224342346, "step": 3004 }, { "clip_ratio/high_max": 0.007038468029350042, "clip_ratio/high_mean": 0.007038468029350042, "clip_ratio/low_mean": 0.009682861273176968, "clip_ratio/low_min": 0.009682861273176968, "clip_ratio/region_mean": 0.01672132930252701, "completions/clipped_ratio": 0.0, "completions/max_length": 197.0, "completions/max_terminated_length": 197.0, "completions/mean_length": 194.5, "completions/mean_terminated_length": 194.5, "completions/min_length": 190.0, "completions/min_terminated_length": 190.0, "entropy": 0.3335018455982208, "epoch": 0.1206972727637868, "frac_reward_zero_std": 0.0, "grad_norm": 1.2387527227401733, "learning_rate": 8.96969696969697e-07, "loss": 0.0018, "num_tokens": 6824278.0, "reward": 0.9990871548652649, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990871548652649, "reward_meter_std": 8.784286910668015e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.781959331827238e-05, "reward_total_composite_mean": 0.9990871548652649, "reward_total_composite_std": 8.784286910668015e-05, "reward_total_mean": 0.9990871548652649, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990871548652649, "rewards/meter/std": 8.784286910668015e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990871548652649, "rewards/total_composite/std": 8.784286910668015e-05, "sampling/importance_sampling_ratio/max": 1.4649758338928223, "sampling/importance_sampling_ratio/mean": 1.0082441568374634, "sampling/importance_sampling_ratio/min": 0.4057033956050873, "sampling/sampling_logp_difference/max": 0.9021329879760742, "sampling/sampling_logp_difference/mean": 0.03491505607962608, "step": 3005 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0011012316681444645, "epoch": 0.12073743824557176, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.93939393939394e-07, "loss": 0.0, "num_tokens": 6825838.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.003272533416748, "sampling/importance_sampling_ratio/mean": 1.0001120567321777, "sampling/importance_sampling_ratio/min": 0.9980201125144958, "sampling/sampling_logp_difference/max": 0.003267202526330948, "sampling/sampling_logp_difference/mean": 0.0001296020782319829, "step": 3006 }, { "clip_ratio/high_max": 0.02804516162723303, "clip_ratio/high_mean": 0.02804516162723303, "clip_ratio/low_mean": 0.006006060168147087, "clip_ratio/low_min": 0.006006060168147087, "clip_ratio/region_mean": 0.034051221795380116, "completions/clipped_ratio": 0.0, "completions/max_length": 173.0, "completions/max_terminated_length": 173.0, "completions/mean_length": 167.875, "completions/mean_terminated_length": 167.875, "completions/min_length": 164.0, "completions/min_terminated_length": 164.0, "entropy": 0.31089969351887703, "epoch": 0.12077760372735671, "frac_reward_zero_std": 0.0, "grad_norm": 3.9469776153564453, "learning_rate": 8.90909090909091e-07, "loss": 0.002, "num_tokens": 6828661.0, "reward": 0.9709138870239258, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986657500267029, "reward_meter_std": 0.0007131828460842371, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05114204064011574, "reward_total_composite_mean": 0.9709138870239258, "reward_total_composite_std": 0.05114205181598663, "reward_total_mean": 0.9709138870239258, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986657500267029, "rewards/meter/std": 0.0007131828460842371, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9709138870239258, "rewards/total_composite/std": 0.05114205181598663, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0033116340637207, "sampling/importance_sampling_ratio/min": 0.09965749830007553, "sampling/sampling_logp_difference/max": 2.306015968322754, "sampling/sampling_logp_difference/mean": 0.04064127802848816, "step": 3007 }, { "clip_ratio/high_max": 0.013912130380049348, "clip_ratio/high_mean": 0.013912130380049348, "clip_ratio/low_mean": 0.008936886792071164, "clip_ratio/low_min": 0.008936886792071164, "clip_ratio/region_mean": 0.02284901717212051, "completions/clipped_ratio": 0.0, "completions/max_length": 391.0, "completions/max_terminated_length": 391.0, "completions/mean_length": 380.375, "completions/mean_terminated_length": 380.375, "completions/min_length": 354.0, "completions/min_terminated_length": 354.0, "entropy": 0.40278755873441696, "epoch": 0.12081776920914167, "frac_reward_zero_std": 0.0, "grad_norm": 1.8470708131790161, "learning_rate": 8.87878787878788e-07, "loss": 0.0014, "num_tokens": 6833680.0, "reward": 0.8320455551147461, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8863636255264282, "reward_count_adherence_std": 0.04208274558186531, "reward_meter_mean": 0.9990596771240234, "reward_meter_std": 0.00017261884931940585, "reward_repeat_penalty_mean": 0.9404239654541016, "reward_repeat_penalty_std": 0.06556747853755951, "reward_std": 0.060938093811273575, "reward_total_composite_mean": 0.8320455551147461, "reward_total_composite_std": 0.06093808636069298, "reward_total_mean": 0.8320455551147461, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8863636255264282, "rewards/count_adherence/std": 0.04208274558186531, "rewards/meter/mean": 0.9990596771240234, "rewards/meter/std": 0.00017261884931940585, "rewards/repeat_penalty/mean": 0.9404239654541016, "rewards/repeat_penalty/std": 0.06556747853755951, "rewards/total_composite/mean": 0.8320455551147461, "rewards/total_composite/std": 0.06093808636069298, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.011826992034912, "sampling/importance_sampling_ratio/min": 0.29510918259620667, "sampling/sampling_logp_difference/max": 1.220409870147705, "sampling/sampling_logp_difference/mean": 0.044413696974515915, "step": 3008 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0017017422651406378, "epoch": 0.12085793469092662, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.84848484848485e-07, "loss": 0.0, "num_tokens": 6835408.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0028958320617676, "sampling/importance_sampling_ratio/mean": 1.000199556350708, "sampling/importance_sampling_ratio/min": 0.9967827200889587, "sampling/sampling_logp_difference/max": 0.0032224380411207676, "sampling/sampling_logp_difference/mean": 0.0002200069575337693, "step": 3009 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0004426717623573495, "epoch": 0.12089810017271158, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.818181818181819e-07, "loss": 0.0, "num_tokens": 6836936.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0006588697433472, "sampling/importance_sampling_ratio/mean": 1.0000369548797607, "sampling/importance_sampling_ratio/min": 0.9992645978927612, "sampling/sampling_logp_difference/max": 0.0007356764399446547, "sampling/sampling_logp_difference/mean": 4.44791694462765e-05, "step": 3010 }, { "clip_ratio/high_max": 0.003494060132652521, "clip_ratio/high_mean": 0.003494060132652521, "clip_ratio/low_mean": 0.002358490601181984, "clip_ratio/low_min": 0.002358490601181984, "clip_ratio/region_mean": 0.005852550733834505, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 106.25, "completions/mean_terminated_length": 106.25, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.13793968316167593, "epoch": 0.12093826565449653, "frac_reward_zero_std": 0.0, "grad_norm": 0.6899365782737732, "learning_rate": 8.787878787878788e-07, "loss": -0.0004, "num_tokens": 6839162.0, "reward": 0.999291181564331, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999291181564331, "reward_meter_std": 7.206718146335334e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.208617898868397e-05, "reward_total_composite_mean": 0.999291181564331, "reward_total_composite_std": 7.206718146335334e-05, "reward_total_mean": 0.999291181564331, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999291181564331, "rewards/meter/std": 7.206718146335334e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999291181564331, "rewards/total_composite/std": 7.206718146335334e-05, "sampling/importance_sampling_ratio/max": 1.7733577489852905, "sampling/importance_sampling_ratio/mean": 1.0045570135116577, "sampling/importance_sampling_ratio/min": 0.3859454393386841, "sampling/sampling_logp_difference/max": 0.952059268951416, "sampling/sampling_logp_difference/mean": 0.013669301755726337, "step": 3011 }, { "clip_ratio/high_max": 0.008762876153923571, "clip_ratio/high_mean": 0.008762876153923571, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.012550755054689944, "completions/clipped_ratio": 0.0, "completions/max_length": 101.0, "completions/max_terminated_length": 101.0, "completions/mean_length": 99.875, "completions/mean_terminated_length": 99.875, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.159406297840178, "epoch": 0.12097843113628148, "frac_reward_zero_std": 0.0, "grad_norm": 3.873194694519043, "learning_rate": 8.757575757575758e-07, "loss": 0.0008, "num_tokens": 6841385.0, "reward": 0.9741964340209961, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991791248321533, "reward_meter_std": 0.00015561260806862265, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07060984522104263, "reward_total_composite_mean": 0.9741964340209961, "reward_total_composite_std": 0.07060983777046204, "reward_total_mean": 0.9741964340209961, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991791248321533, "rewards/meter/std": 0.00015561260806862265, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9741964340209961, "rewards/total_composite/std": 0.07060983777046204, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0067286491394043, "sampling/importance_sampling_ratio/min": 0.07525696605443954, "sampling/sampling_logp_difference/max": 2.5868468284606934, "sampling/sampling_logp_difference/mean": 0.023857776075601578, "step": 3012 }, { "clip_ratio/high_max": 0.00945243111345917, "clip_ratio/high_mean": 0.00945243111345917, "clip_ratio/low_mean": 0.0013440860202535987, "clip_ratio/low_min": 0.0013440860202535987, "clip_ratio/region_mean": 0.010796517133712769, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 92.75, "completions/mean_terminated_length": 92.75, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.06945714261382818, "epoch": 0.12101859661806644, "frac_reward_zero_std": 0.0, "grad_norm": 1.6813957691192627, "learning_rate": 8.727272727272728e-07, "loss": -0.0001, "num_tokens": 6843559.0, "reward": 0.9975481629371643, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975481629371643, "reward_meter_std": 0.0004655694356188178, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00046558206668123603, "reward_total_composite_mean": 0.9975481629371643, "reward_total_composite_std": 0.0004655694356188178, "reward_total_mean": 0.9975481629371643, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975481629371643, "rewards/meter/std": 0.0004655694356188178, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975481629371643, "rewards/total_composite/std": 0.0004655694356188178, "sampling/importance_sampling_ratio/max": 1.3771406412124634, "sampling/importance_sampling_ratio/mean": 1.0000393390655518, "sampling/importance_sampling_ratio/min": 0.3625381588935852, "sampling/sampling_logp_difference/max": 1.0146255493164062, "sampling/sampling_logp_difference/mean": 0.01227449532598257, "step": 3013 }, { "clip_ratio/high_max": 0.007061248528771102, "clip_ratio/high_mean": 0.007061248528771102, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/region_mean": 0.008821811876259744, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 141.625, "completions/mean_terminated_length": 141.625, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.17108981497585773, "epoch": 0.12105876209985139, "frac_reward_zero_std": 0.0, "grad_norm": 1.1659772396087646, "learning_rate": 8.696969696969699e-07, "loss": -0.0004, "num_tokens": 6846028.0, "reward": 0.999117910861969, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999117910861969, "reward_meter_std": 0.00024422150454483926, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002442096301820129, "reward_total_composite_mean": 0.999117910861969, "reward_total_composite_std": 0.00024422150454483926, "reward_total_mean": 0.999117910861969, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999117910861969, "rewards/meter/std": 0.00024422150454483926, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999117910861969, "rewards/total_composite/std": 0.00024422150454483926, "sampling/importance_sampling_ratio/max": 1.4682461023330688, "sampling/importance_sampling_ratio/mean": 1.0043392181396484, "sampling/importance_sampling_ratio/min": 0.47085580229759216, "sampling/sampling_logp_difference/max": 0.7532033920288086, "sampling/sampling_logp_difference/mean": 0.01364456582814455, "step": 3014 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0011554057127796113, "epoch": 0.12109892758163635, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.666666666666668e-07, "loss": 0.0, "num_tokens": 6847788.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0020222663879395, "sampling/importance_sampling_ratio/mean": 1.0000646114349365, "sampling/importance_sampling_ratio/min": 0.9989049434661865, "sampling/sampling_logp_difference/max": 0.0020202400628477335, "sampling/sampling_logp_difference/mean": 9.00184822967276e-05, "step": 3015 }, { "clip_ratio/high_max": 0.02051854378078133, "clip_ratio/high_mean": 0.02051854378078133, "clip_ratio/low_mean": 0.008875739760696888, "clip_ratio/low_min": 0.008875739760696888, "clip_ratio/region_mean": 0.029394283541478217, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 166.375, "completions/mean_terminated_length": 166.375, "completions/min_length": 160.0, "completions/min_terminated_length": 160.0, "entropy": 0.32815663516521454, "epoch": 0.1211390930634213, "frac_reward_zero_std": 0.0, "grad_norm": 3.0696825981140137, "learning_rate": 8.636363636363637e-07, "loss": 0.0068, "num_tokens": 6850815.0, "reward": 0.9988635778427124, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988635778427124, "reward_meter_std": 0.0006956413271836936, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006956506404094398, "reward_total_composite_mean": 0.9988635778427124, "reward_total_composite_std": 0.0006956413271836936, "reward_total_mean": 0.9988635778427124, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988635778427124, "rewards/meter/std": 0.0006956413271836936, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988635778427124, "rewards/total_composite/std": 0.0006956413271836936, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0038132667541504, "sampling/importance_sampling_ratio/min": 0.11682181060314178, "sampling/sampling_logp_difference/max": 2.1471054553985596, "sampling/sampling_logp_difference/mean": 0.04618171229958534, "step": 3016 }, { "clip_ratio/high_max": 0.014189229113981128, "clip_ratio/high_mean": 0.014189229113981128, "clip_ratio/low_mean": 0.0025641699321568012, "clip_ratio/low_min": 0.0025641699321568012, "clip_ratio/region_mean": 0.01675339904613793, "completions/clipped_ratio": 0.0, "completions/max_length": 99.0, "completions/max_terminated_length": 99.0, "completions/mean_length": 97.375, "completions/mean_terminated_length": 97.375, "completions/min_length": 96.0, "completions/min_terminated_length": 96.0, "entropy": 0.22009673342108727, "epoch": 0.12117925854520625, "frac_reward_zero_std": 0.0, "grad_norm": 2.4396045207977295, "learning_rate": 8.606060606060607e-07, "loss": 0.0083, "num_tokens": 6852962.0, "reward": 0.992943286895752, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.992943286895752, "reward_meter_std": 0.0015025768661871552, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0015025660395622253, "reward_total_composite_mean": 0.992943286895752, "reward_total_composite_std": 0.0015025768661871552, "reward_total_mean": 0.992943286895752, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.992943286895752, "rewards/meter/std": 0.0015025768661871552, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.992943286895752, "rewards/total_composite/std": 0.0015025768661871552, "sampling/importance_sampling_ratio/max": 1.6669363975524902, "sampling/importance_sampling_ratio/mean": 1.0036437511444092, "sampling/importance_sampling_ratio/min": 0.41272157430648804, "sampling/sampling_logp_difference/max": 0.8849821090698242, "sampling/sampling_logp_difference/mean": 0.026139099150896072, "step": 3017 }, { "clip_ratio/high_max": 0.016257623909041286, "clip_ratio/high_mean": 0.016257623909041286, "clip_ratio/low_mean": 0.009317906049545854, "clip_ratio/low_min": 0.009317906049545854, "clip_ratio/region_mean": 0.02557552995858714, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 375.75, "completions/mean_terminated_length": 375.75, "completions/min_length": 348.0, "completions/min_terminated_length": 348.0, "entropy": 0.4122394025325775, "epoch": 0.12121942402699121, "frac_reward_zero_std": 0.0, "grad_norm": 1.3273999691009521, "learning_rate": 8.575757575757576e-07, "loss": -0.0211, "num_tokens": 6857392.0, "reward": 0.8386253118515015, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.04704993963241577, "reward_meter_mean": 0.9991037845611572, "reward_meter_std": 9.388383477926254e-05, "reward_repeat_penalty_mean": 0.9593868255615234, "reward_repeat_penalty_std": 0.03774290531873703, "reward_std": 0.05476517230272293, "reward_total_composite_mean": 0.8386253118515015, "reward_total_composite_std": 0.054765164852142334, "reward_total_mean": 0.8386253118515015, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.04704993963241577, "rewards/meter/mean": 0.9991037845611572, "rewards/meter/std": 9.388383477926254e-05, "rewards/repeat_penalty/mean": 0.9593868255615234, "rewards/repeat_penalty/std": 0.03774290531873703, "rewards/total_composite/mean": 0.8386253118515015, "rewards/total_composite/std": 0.054765164852142334, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.012117624282837, "sampling/importance_sampling_ratio/min": 0.2658666670322418, "sampling/sampling_logp_difference/max": 1.3247604370117188, "sampling/sampling_logp_difference/mean": 0.04461507499217987, "step": 3018 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.004098360426723957, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.021813111379742622, "epoch": 0.12125958950877616, "frac_reward_zero_std": 0.0, "grad_norm": 0.24967576563358307, "learning_rate": 8.545454545454546e-07, "loss": -0.0003, "num_tokens": 6859104.0, "reward": 0.9973341226577759, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973341226577759, "reward_meter_std": 9.24646246858174e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.240244253305718e-06, "reward_total_composite_mean": 0.9973341226577759, "reward_total_composite_std": 9.24646246858174e-06, "reward_total_mean": 0.9973341226577759, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973341226577759, "rewards/meter/std": 9.24646246858174e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973341226577759, "rewards/total_composite/std": 9.24646246858174e-06, "sampling/importance_sampling_ratio/max": 1.1754417419433594, "sampling/importance_sampling_ratio/mean": 1.0005959272384644, "sampling/importance_sampling_ratio/min": 0.8715769052505493, "sampling/sampling_logp_difference/max": 0.16164398193359375, "sampling/sampling_logp_difference/mean": 0.002222540322691202, "step": 3019 }, { "clip_ratio/high_max": 0.0025641699321568012, "clip_ratio/high_mean": 0.0025641699321568012, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/region_mean": 0.003839680110104382, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.875, "completions/mean_terminated_length": 97.875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.037544872146099806, "epoch": 0.12129975499056111, "frac_reward_zero_std": 0.0, "grad_norm": 0.15372411906719208, "learning_rate": 8.515151515151515e-07, "loss": 0.0005, "num_tokens": 6861223.0, "reward": 0.9994118809700012, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994118809700012, "reward_meter_std": 1.710540527710691e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.710999640636146e-05, "reward_total_composite_mean": 0.9994118809700012, "reward_total_composite_std": 1.710540527710691e-05, "reward_total_mean": 0.9994118809700012, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994118809700012, "rewards/meter/std": 1.710540527710691e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994118809700012, "rewards/total_composite/std": 1.710540527710691e-05, "sampling/importance_sampling_ratio/max": 1.9928275346755981, "sampling/importance_sampling_ratio/mean": 1.00173020362854, "sampling/importance_sampling_ratio/min": 0.5627803802490234, "sampling/sampling_logp_difference/max": 0.6895544528961182, "sampling/sampling_logp_difference/mean": 0.006246503908187151, "step": 3020 }, { "clip_ratio/high_max": 0.011962095857597888, "clip_ratio/high_mean": 0.011962095857597888, "clip_ratio/low_mean": 0.0021146765793673694, "clip_ratio/low_min": 0.0021146765793673694, "clip_ratio/region_mean": 0.014076772436965257, "completions/clipped_ratio": 0.0, "completions/max_length": 179.0, "completions/max_terminated_length": 179.0, "completions/mean_length": 177.625, "completions/mean_terminated_length": 177.625, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.20351556316018105, "epoch": 0.12133992047234607, "frac_reward_zero_std": 0.0, "grad_norm": 1.7028465270996094, "learning_rate": 8.484848484848486e-07, "loss": 0.0012, "num_tokens": 6864084.0, "reward": 0.9713343381881714, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990875720977783, "reward_meter_std": 0.00012965931091457605, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.05137185752391815, "reward_total_composite_mean": 0.9713343381881714, "reward_total_composite_std": 0.05137185752391815, "reward_total_mean": 0.9713343381881714, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990875720977783, "rewards/meter/std": 0.00012965931091457605, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9713343381881714, "rewards/total_composite/std": 0.05137185752391815, "sampling/importance_sampling_ratio/max": 1.6029589176177979, "sampling/importance_sampling_ratio/mean": 1.004658818244934, "sampling/importance_sampling_ratio/min": 0.405872106552124, "sampling/sampling_logp_difference/max": 0.9017171859741211, "sampling/sampling_logp_difference/mean": 0.018357736989855766, "step": 3021 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.00045082847282174043, "epoch": 0.12138008595413102, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.454545454545456e-07, "loss": 0.0, "num_tokens": 6865556.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0005345344543457, "sampling/importance_sampling_ratio/mean": 1.0000536441802979, "sampling/importance_sampling_ratio/min": 0.9995866417884827, "sampling/sampling_logp_difference/max": 0.0005344097153283656, "sampling/sampling_logp_difference/mean": 5.831335874972865e-05, "step": 3022 }, { "clip_ratio/high_max": 0.017015473917126656, "clip_ratio/high_mean": 0.017015473917126656, "clip_ratio/low_mean": 0.012183514423668385, "clip_ratio/low_min": 0.012183514423668385, "clip_ratio/region_mean": 0.02919898834079504, "completions/clipped_ratio": 0.0, "completions/max_length": 104.0, "completions/max_terminated_length": 104.0, "completions/mean_length": 102.625, "completions/mean_terminated_length": 102.625, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.359358724206686, "epoch": 0.12142025143591598, "frac_reward_zero_std": 0.0, "grad_norm": 3.5771987438201904, "learning_rate": 8.424242424242425e-07, "loss": 0.0024, "num_tokens": 6867729.0, "reward": 0.9883763194084167, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9883763194084167, "reward_meter_std": 0.009592375718057156, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.009592383168637753, "reward_total_composite_mean": 0.9883763194084167, "reward_total_composite_std": 0.009592375718057156, "reward_total_mean": 0.9883763194084167, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9883763194084167, "rewards/meter/std": 0.009592375718057156, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9883763194084167, "rewards/total_composite/std": 0.009592375718057156, "sampling/importance_sampling_ratio/max": 1.92518150806427, "sampling/importance_sampling_ratio/mean": 1.0047192573547363, "sampling/importance_sampling_ratio/min": 0.38989853858947754, "sampling/sampling_logp_difference/max": 0.941868782043457, "sampling/sampling_logp_difference/mean": 0.039184827357530594, "step": 3023 }, { "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0012755101779475808, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.031393167562782764, "epoch": 0.12146041691770093, "frac_reward_zero_std": 0.0, "grad_norm": 0.03466307371854782, "learning_rate": 8.393939393939395e-07, "loss": -0.0002, "num_tokens": 6869889.0, "reward": 0.9994069337844849, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994069337844849, "reward_meter_std": 1.776606950443238e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.775535338310874e-06, "reward_total_composite_mean": 0.9994069337844849, "reward_total_composite_std": 1.776606950443238e-06, "reward_total_mean": 0.9994069337844849, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994069337844849, "rewards/meter/std": 1.776606950443238e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994069337844849, "rewards/total_composite/std": 1.776606950443238e-06, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0014625787734985, "sampling/importance_sampling_ratio/min": 0.4993957281112671, "sampling/sampling_logp_difference/max": 0.844066858291626, "sampling/sampling_logp_difference/mean": 0.005284504033625126, "step": 3024 }, { "clip_ratio/high_max": 0.006628788076341152, "clip_ratio/high_mean": 0.006628788076341152, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006628788076341152, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 131.875, "completions/mean_terminated_length": 131.875, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.0625457102432847, "epoch": 0.12150058239948588, "frac_reward_zero_std": 0.0, "grad_norm": 0.8292710185050964, "learning_rate": 8.363636363636364e-07, "loss": -0.0004, "num_tokens": 6872288.0, "reward": 0.99941086769104, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.99941086769104, "reward_meter_std": 6.445148028433323e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.445525650633499e-05, "reward_total_composite_mean": 0.99941086769104, "reward_total_composite_std": 6.445148028433323e-05, "reward_total_mean": 0.99941086769104, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.99941086769104, "rewards/meter/std": 6.445148028433323e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.99941086769104, "rewards/total_composite/std": 6.445148028433323e-05, "sampling/importance_sampling_ratio/max": 1.33772611618042, "sampling/importance_sampling_ratio/mean": 1.0006563663482666, "sampling/importance_sampling_ratio/min": 0.5258695483207703, "sampling/sampling_logp_difference/max": 0.6427021026611328, "sampling/sampling_logp_difference/mean": 0.008050598204135895, "step": 3025 }, { "clip_ratio/high_max": 0.01498432899825275, "clip_ratio/high_mean": 0.01498432899825275, "clip_ratio/low_mean": 0.023373008705675602, "clip_ratio/low_min": 0.023373008705675602, "clip_ratio/region_mean": 0.03835733770392835, "completions/clipped_ratio": 0.0, "completions/max_length": 127.0, "completions/max_terminated_length": 127.0, "completions/mean_length": 116.25, "completions/mean_terminated_length": 116.25, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.30979629419744015, "epoch": 0.12154074788127084, "frac_reward_zero_std": 0.0, "grad_norm": 6.055728435516357, "learning_rate": 8.333333333333333e-07, "loss": 0.0631, "num_tokens": 6874658.0, "reward": 0.8265341520309448, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.10690449178218842, "reward_meter_mean": 0.9515209197998047, "reward_meter_std": 0.023815272375941277, "reward_repeat_penalty_mean": 0.9633838534355164, "reward_repeat_penalty_std": 0.05091821402311325, "reward_std": 0.12215511500835419, "reward_total_composite_mean": 0.8265341520309448, "reward_total_composite_std": 0.12215512990951538, "reward_total_mean": 0.8265341520309448, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.10690449178218842, "rewards/meter/mean": 0.9515209197998047, "rewards/meter/std": 0.023815272375941277, "rewards/repeat_penalty/mean": 0.9633838534355164, "rewards/repeat_penalty/std": 0.05091821402311325, "rewards/total_composite/mean": 0.8265341520309448, "rewards/total_composite/std": 0.12215512990951538, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9996247291564941, "sampling/importance_sampling_ratio/min": 0.1217365711927414, "sampling/sampling_logp_difference/max": 2.105895757675171, "sampling/sampling_logp_difference/mean": 0.06652670353651047, "step": 3026 }, { "clip_ratio/high_max": 0.014263797434978187, "clip_ratio/high_mean": 0.014263797434978187, "clip_ratio/low_mean": 0.007972249411977828, "clip_ratio/low_min": 0.007972249411977828, "clip_ratio/region_mean": 0.022236046846956015, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 78.625, "completions/mean_terminated_length": 78.625, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.18712522648274899, "epoch": 0.12158091336305579, "frac_reward_zero_std": 0.0, "grad_norm": 1.3954490423202515, "learning_rate": 8.303030303030303e-07, "loss": -0.0003, "num_tokens": 6876567.0, "reward": 0.9990073442459106, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990073442459106, "reward_meter_std": 0.000114113834570162, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011411593004595488, "reward_total_composite_mean": 0.9990073442459106, "reward_total_composite_std": 0.000114113834570162, "reward_total_mean": 0.9990073442459106, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990073442459106, "rewards/meter/std": 0.000114113834570162, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990073442459106, "rewards/total_composite/std": 0.000114113834570162, "sampling/importance_sampling_ratio/max": 1.6200668811798096, "sampling/importance_sampling_ratio/mean": 1.00530207157135, "sampling/importance_sampling_ratio/min": 0.48507970571517944, "sampling/sampling_logp_difference/max": 0.7234420776367188, "sampling/sampling_logp_difference/mean": 0.02237360179424286, "step": 3027 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.015828296658582985, "epoch": 0.12162107884484075, "frac_reward_zero_std": 0.0, "grad_norm": 0.22835104167461395, "learning_rate": 8.272727272727274e-07, "loss": -0.0, "num_tokens": 6878303.0, "reward": 0.9973360300064087, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973360300064087, "reward_meter_std": 8.113272997434251e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.10424353403505e-06, "reward_total_composite_mean": 0.9973360300064087, "reward_total_composite_std": 8.113272997434251e-06, "reward_total_mean": 0.9973360300064087, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973360300064087, "rewards/meter/std": 8.113272997434251e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973360300064087, "rewards/total_composite/std": 8.113272997434251e-06, "sampling/importance_sampling_ratio/max": 1.1025102138519287, "sampling/importance_sampling_ratio/mean": 0.9999331831932068, "sampling/importance_sampling_ratio/min": 0.48051363229751587, "sampling/sampling_logp_difference/max": 0.7328996658325195, "sampling/sampling_logp_difference/mean": 0.0027969391085207462, "step": 3028 }, { "clip_ratio/high_max": 0.008112668758258224, "clip_ratio/high_mean": 0.008112668758258224, "clip_ratio/low_mean": 0.008082556771114469, "clip_ratio/low_min": 0.008082556771114469, "clip_ratio/region_mean": 0.016195225529372692, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 107.875, "completions/mean_terminated_length": 107.875, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.1494951182976365, "epoch": 0.1216612443266257, "frac_reward_zero_std": 0.0, "grad_norm": 0.5074459314346313, "learning_rate": 8.242424242424244e-07, "loss": 0.0018, "num_tokens": 6880686.0, "reward": 0.9992820620536804, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992820620536804, "reward_meter_std": 4.010711199953221e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.012084173155017e-05, "reward_total_composite_mean": 0.9992820620536804, "reward_total_composite_std": 4.010711199953221e-05, "reward_total_mean": 0.9992820620536804, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992820620536804, "rewards/meter/std": 4.010711199953221e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992820620536804, "rewards/total_composite/std": 4.010711199953221e-05, "sampling/importance_sampling_ratio/max": 1.7053853273391724, "sampling/importance_sampling_ratio/mean": 1.0014318227767944, "sampling/importance_sampling_ratio/min": 0.29474636912345886, "sampling/sampling_logp_difference/max": 1.2216401100158691, "sampling/sampling_logp_difference/mean": 0.018780170008540154, "step": 3029 }, { "clip_ratio/high_max": 0.033679026179015636, "clip_ratio/high_mean": 0.033679026179015636, "clip_ratio/low_mean": 0.0044835947919636965, "clip_ratio/low_min": 0.0044835947919636965, "clip_ratio/region_mean": 0.03816262097097933, "completions/clipped_ratio": 0.0, "completions/max_length": 343.0, "completions/max_terminated_length": 343.0, "completions/mean_length": 334.5, "completions/mean_terminated_length": 334.5, "completions/min_length": 323.0, "completions/min_terminated_length": 323.0, "entropy": 0.4611641392111778, "epoch": 0.12170140980841065, "frac_reward_zero_std": 0.0, "grad_norm": 1.8386452198028564, "learning_rate": 8.212121212121213e-07, "loss": -0.0054, "num_tokens": 6885074.0, "reward": 0.8695409297943115, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9090909361839294, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9939321279525757, "reward_meter_std": 0.0059409006498754025, "reward_repeat_penalty_mean": 0.9621710777282715, "reward_repeat_penalty_std": 0.08768600970506668, "reward_std": 0.08088699728250504, "reward_total_composite_mean": 0.8695409297943115, "reward_total_composite_std": 0.08088699728250504, "reward_total_mean": 0.8695409297943115, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9090909361839294, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9939321279525757, "rewards/meter/std": 0.0059409006498754025, "rewards/repeat_penalty/mean": 0.9621710777282715, "rewards/repeat_penalty/std": 0.08768600970506668, "rewards/total_composite/mean": 0.8695409297943115, "rewards/total_composite/std": 0.08088699728250504, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0112972259521484, "sampling/importance_sampling_ratio/min": 0.05401458963751793, "sampling/sampling_logp_difference/max": 2.918501138687134, "sampling/sampling_logp_difference/mean": 0.054148994386196136, "step": 3030 }, { "clip_ratio/high_max": 0.02090477745514363, "clip_ratio/high_mean": 0.02090477745514363, "clip_ratio/low_mean": 0.0033333334140479565, "clip_ratio/low_min": 0.0033333334140479565, "clip_ratio/region_mean": 0.024238110869191587, "completions/clipped_ratio": 0.0, "completions/max_length": 158.0, "completions/max_terminated_length": 158.0, "completions/mean_length": 154.75, "completions/mean_terminated_length": 154.75, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "entropy": 0.33194397762417793, "epoch": 0.12174157529019561, "frac_reward_zero_std": 0.0, "grad_norm": 2.3149688243865967, "learning_rate": 8.181818181818182e-07, "loss": -0.0067, "num_tokens": 6887904.0, "reward": 0.9982709884643555, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9982709884643555, "reward_meter_std": 0.0024356048088520765, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0024356169160455465, "reward_total_composite_mean": 0.9982709884643555, "reward_total_composite_std": 0.0024356048088520765, "reward_total_mean": 0.9982709884643555, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9982709884643555, "rewards/meter/std": 0.0024356048088520765, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9982709884643555, "rewards/total_composite/std": 0.0024356048088520765, "sampling/importance_sampling_ratio/max": 1.942898154258728, "sampling/importance_sampling_ratio/mean": 1.009082555770874, "sampling/importance_sampling_ratio/min": 0.24117796123027802, "sampling/sampling_logp_difference/max": 1.422220230102539, "sampling/sampling_logp_difference/mean": 0.03701268509030342, "step": 3031 }, { "clip_ratio/high_max": 0.007490954245440662, "clip_ratio/high_mean": 0.007490954245440662, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.009384893695823848, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.05587592348456383, "epoch": 0.12178174077198056, "frac_reward_zero_std": 0.0, "grad_norm": 0.31778454780578613, "learning_rate": 8.151515151515152e-07, "loss": -0.0001, "num_tokens": 6889852.0, "reward": 0.9981399774551392, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981399774551392, "reward_meter_std": 1.4936525076336693e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.4939930224500131e-05, "reward_total_composite_mean": 0.9981399774551392, "reward_total_composite_std": 1.4936525076336693e-05, "reward_total_mean": 0.9981399774551392, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981399774551392, "rewards/meter/std": 1.4936525076336693e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981399774551392, "rewards/total_composite/std": 1.4936525076336693e-05, "sampling/importance_sampling_ratio/max": 1.3944506645202637, "sampling/importance_sampling_ratio/mean": 0.9997544884681702, "sampling/importance_sampling_ratio/min": 0.4725368618965149, "sampling/sampling_logp_difference/max": 0.7496395111083984, "sampling/sampling_logp_difference/mean": 0.007410375867038965, "step": 3032 }, { "clip_ratio/high_max": 0.016598281217738986, "clip_ratio/high_mean": 0.016598281217738986, "clip_ratio/low_mean": 0.006038419320248067, "clip_ratio/low_min": 0.006038419320248067, "clip_ratio/region_mean": 0.022636700537987053, "completions/clipped_ratio": 0.0, "completions/max_length": 189.0, "completions/max_terminated_length": 189.0, "completions/mean_length": 186.875, "completions/mean_terminated_length": 186.875, "completions/min_length": 183.0, "completions/min_terminated_length": 183.0, "entropy": 0.36450883373618126, "epoch": 0.12182190625376552, "frac_reward_zero_std": 0.0, "grad_norm": 2.056260347366333, "learning_rate": 8.121212121212121e-07, "loss": 0.0028, "num_tokens": 6892787.0, "reward": 0.8737800121307373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9618879556655884, "reward_meter_std": 0.06856533139944077, "reward_repeat_penalty_mean": 0.9090909361839294, "reward_repeat_penalty_std": 0.0971859022974968, "reward_std": 0.10824880748987198, "reward_total_composite_mean": 0.8737800121307373, "reward_total_composite_std": 0.10824880003929138, "reward_total_mean": 0.8737800121307373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9618879556655884, "rewards/meter/std": 0.06856533139944077, "rewards/repeat_penalty/mean": 0.9090909361839294, "rewards/repeat_penalty/std": 0.0971859022974968, "rewards/total_composite/mean": 0.8737800121307373, "rewards/total_composite/std": 0.10824880003929138, "sampling/importance_sampling_ratio/max": 1.8432459831237793, "sampling/importance_sampling_ratio/mean": 1.0074325799942017, "sampling/importance_sampling_ratio/min": 0.33824771642684937, "sampling/sampling_logp_difference/max": 1.0839767456054688, "sampling/sampling_logp_difference/mean": 0.034711167216300964, "step": 3033 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.0012948041403433308, "epoch": 0.12186207173555047, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.09090909090909e-07, "loss": 0.0, "num_tokens": 6894235.0, "reward": 0.9943599104881287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9943599104881287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0041109323501587, "sampling/importance_sampling_ratio/mean": 1.000151515007019, "sampling/importance_sampling_ratio/min": 0.9988238215446472, "sampling/sampling_logp_difference/max": 0.004102423787117004, "sampling/sampling_logp_difference/mean": 0.0001638415560591966, "step": 3034 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0020439386717043817, "epoch": 0.12190223721733542, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.060606060606062e-07, "loss": 0.0, "num_tokens": 6895891.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0033022165298462, "sampling/importance_sampling_ratio/mean": 1.0002354383468628, "sampling/importance_sampling_ratio/min": 0.9998509287834167, "sampling/sampling_logp_difference/max": 0.0032967175357043743, "sampling/sampling_logp_difference/mean": 0.0002374651812715456, "step": 3035 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.0024824678112054244, "epoch": 0.12194240269912038, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.030303030303031e-07, "loss": 0.0, "num_tokens": 6897435.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0082646608352661, "sampling/importance_sampling_ratio/mean": 1.0001778602600098, "sampling/importance_sampling_ratio/min": 0.9977694749832153, "sampling/sampling_logp_difference/max": 0.008230682462453842, "sampling/sampling_logp_difference/mean": 0.00020185003813821822, "step": 3036 }, { "clip_ratio/high_max": 0.023894709534943104, "clip_ratio/high_mean": 0.023894709534943104, "clip_ratio/low_mean": 0.0012254902394488454, "clip_ratio/low_min": 0.0012254902394488454, "clip_ratio/region_mean": 0.02512019977439195, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 104.0, "completions/mean_terminated_length": 104.0, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.2963132858276367, "epoch": 0.12198256818090533, "frac_reward_zero_std": 0.0, "grad_norm": 4.432851791381836, "learning_rate": 8.000000000000001e-07, "loss": 0.0, "num_tokens": 6899483.0, "reward": 0.9615679979324341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9864717721939087, "reward_meter_std": 0.024548275396227837, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07080353796482086, "reward_total_composite_mean": 0.9615679979324341, "reward_total_composite_std": 0.07080353796482086, "reward_total_mean": 0.9615679979324341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9864717721939087, "rewards/meter/std": 0.024548275396227837, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9615679979324341, "rewards/total_composite/std": 0.07080353796482086, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0080146789550781, "sampling/importance_sampling_ratio/min": 0.2869689166545868, "sampling/sampling_logp_difference/max": 1.2483813762664795, "sampling/sampling_logp_difference/mean": 0.03678695112466812, "step": 3037 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.12202273366269029, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 7.96969696969697e-07, "loss": 0.0, "num_tokens": 6900995.0, "reward": 0.6106069087982178, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6749999523162842, "reward_count_adherence_std": 0.02672613225877285, "reward_meter_mean": 0.933377742767334, "reward_meter_std": 0.09035933762788773, "reward_repeat_penalty_mean": 0.9716880321502686, "reward_repeat_penalty_std": 0.01748696342110634, "reward_std": 0.045595090836286545, "reward_total_composite_mean": 0.6106069087982178, "reward_total_composite_std": 0.045595090836286545, "reward_total_mean": 0.6106069087982178, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6749999523162842, "rewards/count_adherence/std": 0.02672613225877285, "rewards/meter/mean": 0.933377742767334, "rewards/meter/std": 0.09035933762788773, "rewards/repeat_penalty/mean": 0.9716880321502686, "rewards/repeat_penalty/std": 0.01748696342110634, "rewards/total_composite/mean": 0.6106069087982178, "rewards/total_composite/std": 0.045595090836286545, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 3038 }, { "clip_ratio/high_max": 0.022452297853305936, "clip_ratio/high_mean": 0.022452297853305936, "clip_ratio/low_mean": 0.0021551724057644606, "clip_ratio/low_min": 0.0021551724057644606, "clip_ratio/region_mean": 0.024607470259070396, "completions/clipped_ratio": 0.0, "completions/max_length": 238.0, "completions/max_terminated_length": 238.0, "completions/mean_length": 233.25, "completions/mean_terminated_length": 233.25, "completions/min_length": 230.0, "completions/min_terminated_length": 230.0, "entropy": 0.3743705749511719, "epoch": 0.12206289914447524, "frac_reward_zero_std": 0.0, "grad_norm": 1.2539805173873901, "learning_rate": 7.939393939393939e-07, "loss": 0.0015, "num_tokens": 6904381.0, "reward": 0.987464189529419, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988183975219727, "reward_meter_std": 0.000870219839271158, "reward_repeat_penalty_mean": 0.9886363744735718, "reward_repeat_penalty_std": 0.03214120864868164, "reward_std": 0.031983714550733566, "reward_total_composite_mean": 0.987464189529419, "reward_total_composite_std": 0.03198371082544327, "reward_total_mean": 0.987464189529419, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988183975219727, "rewards/meter/std": 0.000870219839271158, "rewards/repeat_penalty/mean": 0.9886363744735718, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.987464189529419, "rewards/total_composite/std": 0.03198371082544327, "sampling/importance_sampling_ratio/max": 1.7767890691757202, "sampling/importance_sampling_ratio/mean": 1.012012243270874, "sampling/importance_sampling_ratio/min": 0.25801458954811096, "sampling/sampling_logp_difference/max": 1.3547391891479492, "sampling/sampling_logp_difference/mean": 0.03573669493198395, "step": 3039 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.00237001696950756, "epoch": 0.1221030646262602, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.909090909090909e-07, "loss": 0.0, "num_tokens": 6906301.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0078413486480713, "sampling/importance_sampling_ratio/mean": 1.000292420387268, "sampling/importance_sampling_ratio/min": 0.9997765421867371, "sampling/sampling_logp_difference/max": 0.007810796611011028, "sampling/sampling_logp_difference/mean": 0.0002939449332188815, "step": 3040 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.0016474624135298654, "epoch": 0.12214323010804515, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.878787878787879e-07, "loss": 0.0, "num_tokens": 6907797.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0073140859603882, "sampling/importance_sampling_ratio/mean": 1.0001239776611328, "sampling/importance_sampling_ratio/min": 0.9921550154685974, "sampling/sampling_logp_difference/max": 0.007875919342041016, "sampling/sampling_logp_difference/mean": 0.00019611754396464676, "step": 3041 }, { "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/low_mean": 0.0034966744715347886, "clip_ratio/low_min": 0.0034966744715347886, "clip_ratio/region_mean": 0.00525723781902343, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.125, "completions/mean_terminated_length": 71.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.0695438552647829, "epoch": 0.1221833955898301, "frac_reward_zero_std": 0.0, "grad_norm": 0.7192491292953491, "learning_rate": 7.84848484848485e-07, "loss": -0.0003, "num_tokens": 6909726.0, "reward": 0.9994059801101685, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994059801101685, "reward_meter_std": 3.530035974108614e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.531122638378292e-05, "reward_total_composite_mean": 0.9994059801101685, "reward_total_composite_std": 3.530035974108614e-05, "reward_total_mean": 0.9994059801101685, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994059801101685, "rewards/meter/std": 3.530035974108614e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994059801101685, "rewards/total_composite/std": 3.530035974108614e-05, "sampling/importance_sampling_ratio/max": 1.1902652978897095, "sampling/importance_sampling_ratio/mean": 1.0006753206253052, "sampling/importance_sampling_ratio/min": 0.39109766483306885, "sampling/sampling_logp_difference/max": 0.9387979507446289, "sampling/sampling_logp_difference/mean": 0.008167088031768799, "step": 3042 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0016416620055679232, "epoch": 0.12222356107161506, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.818181818181819e-07, "loss": 0.0, "num_tokens": 6911286.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0025135278701782, "sampling/importance_sampling_ratio/mean": 1.0002003908157349, "sampling/importance_sampling_ratio/min": 0.9998286962509155, "sampling/sampling_logp_difference/max": 0.002510334365069866, "sampling/sampling_logp_difference/mean": 0.0002015612117247656, "step": 3043 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0007720192006672733, "epoch": 0.12226372655340001, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.787878787878788e-07, "loss": 0.0, "num_tokens": 6913358.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0022180080413818, "sampling/importance_sampling_ratio/mean": 1.0000611543655396, "sampling/importance_sampling_ratio/min": 0.9950957298278809, "sampling/sampling_logp_difference/max": 0.004916388541460037, "sampling/sampling_logp_difference/mean": 0.00012283638352528214, "step": 3044 }, { "clip_ratio/high_max": 0.01052544778212905, "clip_ratio/high_mean": 0.01052544778212905, "clip_ratio/low_mean": 0.005886243423447013, "clip_ratio/low_min": 0.005886243423447013, "clip_ratio/region_mean": 0.016411691205576062, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 107.0, "completions/mean_terminated_length": 107.0, "completions/min_length": 105.0, "completions/min_terminated_length": 105.0, "entropy": 0.13631522469222546, "epoch": 0.12230389203518496, "frac_reward_zero_std": 0.0, "grad_norm": 1.1839877367019653, "learning_rate": 7.757575757575758e-07, "loss": -0.0027, "num_tokens": 6915686.0, "reward": 0.9992265701293945, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992265701293945, "reward_meter_std": 0.0002075855591101572, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020759036124218255, "reward_total_composite_mean": 0.9992265701293945, "reward_total_composite_std": 0.0002075855591101572, "reward_total_mean": 0.9992265701293945, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992265701293945, "rewards/meter/std": 0.0002075855591101572, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992265701293945, "rewards/total_composite/std": 0.0002075855591101572, "sampling/importance_sampling_ratio/max": 1.8381428718566895, "sampling/importance_sampling_ratio/mean": 1.005001187324524, "sampling/importance_sampling_ratio/min": 0.33878862857818604, "sampling/sampling_logp_difference/max": 1.08237886428833, "sampling/sampling_logp_difference/mean": 0.01808370277285576, "step": 3045 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.0036764706019312143, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 67.875, "completions/mean_terminated_length": 67.875, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.04532818449661136, "epoch": 0.12234405751696992, "frac_reward_zero_std": 0.0, "grad_norm": 1.057833194732666, "learning_rate": 7.727272727272727e-07, "loss": -0.0018, "num_tokens": 6917493.0, "reward": 0.9994407892227173, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994407892227173, "reward_meter_std": 6.79898148518987e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.798980029998347e-05, "reward_total_composite_mean": 0.9994407892227173, "reward_total_composite_std": 6.79898148518987e-05, "reward_total_mean": 0.9994407892227173, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994407892227173, "rewards/meter/std": 6.79898148518987e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994407892227173, "rewards/total_composite/std": 6.79898148518987e-05, "sampling/importance_sampling_ratio/max": 1.2558224201202393, "sampling/importance_sampling_ratio/mean": 1.0019086599349976, "sampling/importance_sampling_ratio/min": 0.7700556516647339, "sampling/sampling_logp_difference/max": 0.2612924575805664, "sampling/sampling_logp_difference/mean": 0.0057174162939190865, "step": 3046 }, { "clip_ratio/high_max": 0.005637656955514103, "clip_ratio/high_mean": 0.005637656955514103, "clip_ratio/low_mean": 0.005625279794912785, "clip_ratio/low_min": 0.005625279794912785, "clip_ratio/region_mean": 0.011262936750426888, "completions/clipped_ratio": 0.0, "completions/max_length": 202.0, "completions/max_terminated_length": 202.0, "completions/mean_length": 200.0, "completions/mean_terminated_length": 200.0, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "entropy": 0.17142035998404026, "epoch": 0.12238422299875487, "frac_reward_zero_std": 0.0, "grad_norm": 1.7859693765640259, "learning_rate": 7.696969696969698e-07, "loss": 0.005, "num_tokens": 6920501.0, "reward": 0.9083393812179565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991546869277954, "reward_meter_std": 0.0004873749567195773, "reward_repeat_penalty_mean": 0.9090908765792847, "reward_repeat_penalty_std": 0.06872081756591797, "reward_std": 0.06891340017318726, "reward_total_composite_mean": 0.9083393812179565, "reward_total_composite_std": 0.06891340762376785, "reward_total_mean": 0.9083393812179565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991546869277954, "rewards/meter/std": 0.0004873749567195773, "rewards/repeat_penalty/mean": 0.9090908765792847, "rewards/repeat_penalty/std": 0.06872081756591797, "rewards/total_composite/mean": 0.9083393812179565, "rewards/total_composite/std": 0.06891340762376785, "sampling/importance_sampling_ratio/max": 1.687934398651123, "sampling/importance_sampling_ratio/mean": 1.0054200887680054, "sampling/importance_sampling_ratio/min": 0.2626442015171051, "sampling/sampling_logp_difference/max": 1.3369550704956055, "sampling/sampling_logp_difference/mean": 0.018795771524310112, "step": 3047 }, { "clip_ratio/high_max": 0.0035714285913854837, "clip_ratio/high_mean": 0.0035714285913854837, "clip_ratio/low_mean": 0.0035714285913854837, "clip_ratio/low_min": 0.0035714285913854837, "clip_ratio/region_mean": 0.0071428571827709675, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 34.25, "completions/mean_terminated_length": 34.25, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.09313629940152168, "epoch": 0.12242438848053983, "frac_reward_zero_std": 0.0, "grad_norm": 5.802453994750977, "learning_rate": 7.666666666666667e-07, "loss": -0.0015, "num_tokens": 6921887.0, "reward": 0.9926164150238037, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9926164150238037, "reward_meter_std": 0.0012579105095937848, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001257919822819531, "reward_total_composite_mean": 0.9926164150238037, "reward_total_composite_std": 0.0012579105095937848, "reward_total_mean": 0.9926164150238037, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9926164150238037, "rewards/meter/std": 0.0012579105095937848, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926164150238037, "rewards/total_composite/std": 0.0012579105095937848, "sampling/importance_sampling_ratio/max": 1.4339221715927124, "sampling/importance_sampling_ratio/mean": 0.9971778988838196, "sampling/importance_sampling_ratio/min": 0.325217068195343, "sampling/sampling_logp_difference/max": 1.1232624053955078, "sampling/sampling_logp_difference/mean": 0.014383560046553612, "step": 3048 }, { "clip_ratio/high_max": 0.01253445539623499, "clip_ratio/high_mean": 0.01253445539623499, "clip_ratio/low_mean": 0.011909950990229845, "clip_ratio/low_min": 0.011909950990229845, "clip_ratio/region_mean": 0.024444406386464834, "completions/clipped_ratio": 0.0, "completions/max_length": 141.0, "completions/max_terminated_length": 141.0, "completions/mean_length": 137.625, "completions/mean_terminated_length": 137.625, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.2945951074361801, "epoch": 0.12246455396232478, "frac_reward_zero_std": 0.0, "grad_norm": 3.5305817127227783, "learning_rate": 7.636363636363637e-07, "loss": -0.0064, "num_tokens": 6924412.0, "reward": 0.8786302804946899, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9496752023696899, "reward_meter_std": 0.0848766416311264, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.0753149539232254, "reward_total_composite_mean": 0.8786302804946899, "reward_total_composite_std": 0.0753149688243866, "reward_total_mean": 0.8786302804946899, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9496752023696899, "rewards/meter/std": 0.0848766416311264, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8786302804946899, "rewards/total_composite/std": 0.0753149688243866, "sampling/importance_sampling_ratio/max": 1.5370413064956665, "sampling/importance_sampling_ratio/mean": 1.0010079145431519, "sampling/importance_sampling_ratio/min": 0.3069687485694885, "sampling/sampling_logp_difference/max": 1.181009292602539, "sampling/sampling_logp_difference/mean": 0.03166425973176956, "step": 3049 }, { "clip_ratio/high_max": 0.016509592533111572, "clip_ratio/high_mean": 0.016509592533111572, "clip_ratio/low_mean": 0.004255396313965321, "clip_ratio/low_min": 0.004255396313965321, "clip_ratio/region_mean": 0.020764988847076893, "completions/clipped_ratio": 0.0, "completions/max_length": 238.0, "completions/max_terminated_length": 238.0, "completions/mean_length": 234.0, "completions/mean_terminated_length": 234.0, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.3369157761335373, "epoch": 0.12250471944410973, "frac_reward_zero_std": 0.0, "grad_norm": 1.4543901681900024, "learning_rate": 7.606060606060607e-07, "loss": -0.0011, "num_tokens": 6928012.0, "reward": 0.9989113807678223, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989113807678223, "reward_meter_std": 0.0005675117135979235, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005675173015333712, "reward_total_composite_mean": 0.9989113807678223, "reward_total_composite_std": 0.0005675117135979235, "reward_total_mean": 0.9989113807678223, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989113807678223, "rewards/meter/std": 0.0005675117135979235, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989113807678223, "rewards/total_composite/std": 0.0005675117135979235, "sampling/importance_sampling_ratio/max": 1.7903379201889038, "sampling/importance_sampling_ratio/mean": 1.007856011390686, "sampling/importance_sampling_ratio/min": 0.22545784711837769, "sampling/sampling_logp_difference/max": 1.4896221160888672, "sampling/sampling_logp_difference/mean": 0.03521440923213959, "step": 3050 }, { "epoch": 0.12250471944410973, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/max_length": 416.3076923076923, "eval_completions/max_terminated_length": 395.53846153846155, "eval_completions/mean_length": 212.6346153846154, "eval_completions/mean_terminated_length": 206.1826934814453, "eval_completions/min_length": 60.76923076923077, "eval_completions/min_terminated_length": 60.76923076923077, "eval_entropy": 0.34184010556110966, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 6928012.0, "eval_reward": 0.722142334167774, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.9593305908716642, "eval_reward_count_adherence_std": 0.06269322364376141, "eval_reward_meter_mean": 0.7826263767022353, "eval_reward_meter_std": 0.3530325018442594, "eval_reward_repeat_penalty_mean": 0.9508223029283377, "eval_reward_repeat_penalty_std": 0.06783326672246823, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.722142334167774, "eval_reward_total_composite_std": 0.3449985155692467, "eval_reward_total_mean": 0.722142334167774, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.9593305908716642, "eval_rewards/count_adherence/std": 0.06269322364376141, "eval_rewards/meter/mean": 0.7826263767022353, "eval_rewards/meter/std": 0.3530325018442594, "eval_rewards/repeat_penalty/mean": 0.9508223029283377, "eval_rewards/repeat_penalty/std": 0.06783326672246823, "eval_rewards/total_composite/mean": 0.722142334167774, "eval_rewards/total_composite/std": 0.3449985155692467, "eval_runtime": 77.5007, "eval_samples_per_second": 1.342, "eval_sampling/importance_sampling_ratio/max": 1.5470959773430457, "eval_sampling/importance_sampling_ratio/mean": 1.0082989380909846, "eval_sampling/importance_sampling_ratio/min": 0.3223212957382202, "eval_sampling/sampling_logp_difference/max": 1.1543501340425932, "eval_sampling/sampling_logp_difference/mean": 0.030635818409231994, "eval_steps_per_second": 0.168, "step": 3050 }, { "clip_ratio/high_max": 0.02257009909953922, "clip_ratio/high_mean": 0.02257009909953922, "clip_ratio/low_mean": 0.0026709402445703745, "clip_ratio/low_min": 0.0026709402445703745, "clip_ratio/region_mean": 0.025241039344109595, "completions/clipped_ratio": 0.0, "completions/max_length": 235.0, "completions/max_terminated_length": 235.0, "completions/mean_length": 233.0, "completions/mean_terminated_length": 233.0, "completions/min_length": 231.0, "completions/min_terminated_length": 231.0, "entropy": 0.3387974835932255, "epoch": 0.12254488492589469, "frac_reward_zero_std": 0.0, "grad_norm": 1.5024135112762451, "learning_rate": 7.575757575757576e-07, "loss": 0.0034, "num_tokens": 6931580.0, "reward": 0.9878134727478027, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999167799949646, "reward_meter_std": 7.227841706480831e-05, "reward_repeat_penalty_mean": 0.9886363744735718, "reward_repeat_penalty_std": 0.03214120864868164, "reward_std": 0.03211057186126709, "reward_total_composite_mean": 0.9878134727478027, "reward_total_composite_std": 0.03211057558655739, "reward_total_mean": 0.9878134727478027, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999167799949646, "rewards/meter/std": 7.227841706480831e-05, "rewards/repeat_penalty/mean": 0.9886363744735718, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.9878134727478027, "rewards/total_composite/std": 0.03211057558655739, "sampling/importance_sampling_ratio/max": 1.6730788946151733, "sampling/importance_sampling_ratio/mean": 1.0094155073165894, "sampling/importance_sampling_ratio/min": 0.29337432980537415, "sampling/sampling_logp_difference/max": 1.2263059616088867, "sampling/sampling_logp_difference/mean": 0.037654075771570206, "step": 3051 }, { "clip_ratio/high_max": 0.016292808344587684, "clip_ratio/high_mean": 0.016292808344587684, "clip_ratio/low_mean": 0.009418385569006205, "clip_ratio/low_min": 0.009418385569006205, "clip_ratio/region_mean": 0.02571119391359389, "completions/clipped_ratio": 0.0, "completions/max_length": 356.0, "completions/max_terminated_length": 356.0, "completions/mean_length": 322.75, "completions/mean_terminated_length": 322.75, "completions/min_length": 308.0, "completions/min_terminated_length": 308.0, "entropy": 0.41726846620440483, "epoch": 0.12258505040767964, "frac_reward_zero_std": 0.0, "grad_norm": 2.0073533058166504, "learning_rate": 7.545454545454546e-07, "loss": 0.0461, "num_tokens": 6935794.0, "reward": 0.9595891237258911, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.9991399645805359, "reward_meter_std": 0.00012341544788796455, "reward_repeat_penalty_mean": 0.9916666746139526, "reward_repeat_penalty_std": 0.0235702246427536, "reward_std": 0.057442426681518555, "reward_total_composite_mean": 0.9595891237258911, "reward_total_composite_std": 0.05744243040680885, "reward_total_mean": 0.9595891237258911, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.9991399645805359, "rewards/meter/std": 0.00012341544788796455, "rewards/repeat_penalty/mean": 0.9916666746139526, "rewards/repeat_penalty/std": 0.0235702246427536, "rewards/total_composite/mean": 0.9595891237258911, "rewards/total_composite/std": 0.05744243040680885, "sampling/importance_sampling_ratio/max": 1.883813500404358, "sampling/importance_sampling_ratio/mean": 1.0115916728973389, "sampling/importance_sampling_ratio/min": 0.2867130637168884, "sampling/sampling_logp_difference/max": 1.2492733001708984, "sampling/sampling_logp_difference/mean": 0.04169188067317009, "step": 3052 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.013010540511459112, "clip_ratio/low_min": 0.013010540511459112, "clip_ratio/region_mean": 0.013010540511459112, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 66.125, "completions/mean_terminated_length": 66.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.10556314932182431, "epoch": 0.1226252158894646, "frac_reward_zero_std": 0.0, "grad_norm": 2.259249448776245, "learning_rate": 7.515151515151516e-07, "loss": 0.0049, "num_tokens": 6937691.0, "reward": 0.9928057193756104, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9928057193756104, "reward_meter_std": 0.0009291421738453209, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009291406604461372, "reward_total_composite_mean": 0.9928057193756104, "reward_total_composite_std": 0.0009291421738453209, "reward_total_mean": 0.9928057193756104, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9928057193756104, "rewards/meter/std": 0.0009291421738453209, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9928057193756104, "rewards/total_composite/std": 0.0009291421738453209, "sampling/importance_sampling_ratio/max": 1.8138948678970337, "sampling/importance_sampling_ratio/mean": 1.0060573816299438, "sampling/importance_sampling_ratio/min": 0.36685484647750854, "sampling/sampling_logp_difference/max": 1.00278902053833, "sampling/sampling_logp_difference/mean": 0.015979096293449402, "step": 3053 }, { "clip_ratio/high_max": 0.026310671470128, "clip_ratio/high_mean": 0.026310671470128, "clip_ratio/low_mean": 0.0018115942366421223, "clip_ratio/low_min": 0.0018115942366421223, "clip_ratio/region_mean": 0.028122265706770122, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 137.875, "completions/mean_terminated_length": 137.875, "completions/min_length": 135.0, "completions/min_terminated_length": 135.0, "entropy": 0.28184244222939014, "epoch": 0.12266538137124955, "frac_reward_zero_std": 0.0, "grad_norm": 2.697394847869873, "learning_rate": 7.484848484848485e-07, "loss": 0.0032, "num_tokens": 6940138.0, "reward": 0.9745643138885498, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9923188090324402, "reward_meter_std": 0.004795641638338566, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.04966310039162636, "reward_total_composite_mean": 0.9745643138885498, "reward_total_composite_std": 0.04966309294104576, "reward_total_mean": 0.9745643138885498, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9923188090324402, "rewards/meter/std": 0.004795641638338566, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9745643138885498, "rewards/total_composite/std": 0.04966309294104576, "sampling/importance_sampling_ratio/max": 1.498881459236145, "sampling/importance_sampling_ratio/mean": 1.0037504434585571, "sampling/importance_sampling_ratio/min": 0.2922235131263733, "sampling/sampling_logp_difference/max": 1.230236291885376, "sampling/sampling_logp_difference/mean": 0.02996024861931801, "step": 3054 }, { "clip_ratio/high_max": 0.025692227762192488, "clip_ratio/high_mean": 0.025692227762192488, "clip_ratio/low_mean": 0.010666976682841778, "clip_ratio/low_min": 0.010666976682841778, "clip_ratio/region_mean": 0.036359204445034266, "completions/clipped_ratio": 0.0, "completions/max_length": 238.0, "completions/max_terminated_length": 238.0, "completions/mean_length": 233.625, "completions/mean_terminated_length": 233.625, "completions/min_length": 228.0, "completions/min_terminated_length": 228.0, "entropy": 0.3986949659883976, "epoch": 0.1227055468530345, "frac_reward_zero_std": 0.0, "grad_norm": 2.5886197090148926, "learning_rate": 7.454545454545455e-07, "loss": 0.0153, "num_tokens": 6943559.0, "reward": 0.9304856061935425, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977272748947144, "reward_meter_std": 0.0019398098811507225, "reward_repeat_penalty_mean": 0.932692289352417, "reward_repeat_penalty_std": 0.08661473542451859, "reward_std": 0.08541877567768097, "reward_total_composite_mean": 0.9304856061935425, "reward_total_composite_std": 0.08541877567768097, "reward_total_mean": 0.9304856061935425, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977272748947144, "rewards/meter/std": 0.0019398098811507225, "rewards/repeat_penalty/mean": 0.932692289352417, "rewards/repeat_penalty/std": 0.08661473542451859, "rewards/total_composite/mean": 0.9304856061935425, "rewards/total_composite/std": 0.08541877567768097, "sampling/importance_sampling_ratio/max": 1.9301133155822754, "sampling/importance_sampling_ratio/mean": 1.0053743124008179, "sampling/importance_sampling_ratio/min": 0.19072523713111877, "sampling/sampling_logp_difference/max": 1.65692138671875, "sampling/sampling_logp_difference/mean": 0.045273296535015106, "step": 3055 }, { "clip_ratio/high_max": 0.01225746376439929, "clip_ratio/high_mean": 0.01225746376439929, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/region_mean": 0.014018027111887932, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 142.25, "completions/mean_terminated_length": 142.25, "completions/min_length": 138.0, "completions/min_terminated_length": 138.0, "entropy": 0.18231415934860706, "epoch": 0.12274571233481946, "frac_reward_zero_std": 0.0, "grad_norm": 1.2885347604751587, "learning_rate": 7.424242424242425e-07, "loss": -0.0003, "num_tokens": 6946121.0, "reward": 0.9813523292541504, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991947412490845, "reward_meter_std": 7.931316940812394e-05, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05047507956624031, "reward_total_composite_mean": 0.9813523292541504, "reward_total_composite_std": 0.050475094467401505, "reward_total_mean": 0.9813523292541504, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991947412490845, "rewards/meter/std": 7.931316940812394e-05, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9813523292541504, "rewards/total_composite/std": 0.050475094467401505, "sampling/importance_sampling_ratio/max": 1.6997216939926147, "sampling/importance_sampling_ratio/mean": 1.004903793334961, "sampling/importance_sampling_ratio/min": 0.4428071081638336, "sampling/sampling_logp_difference/max": 0.8146209716796875, "sampling/sampling_logp_difference/mean": 0.015199076384305954, "step": 3056 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0056535504991188645, "clip_ratio/low_min": 0.0056535504991188645, "clip_ratio/region_mean": 0.0056535504991188645, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.05063098296523094, "epoch": 0.12278587781660441, "frac_reward_zero_std": 0.0, "grad_norm": 0.10690882056951523, "learning_rate": 7.393939393939395e-07, "loss": -0.0002, "num_tokens": 6948039.0, "reward": 0.9981523752212524, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981523752212524, "reward_meter_std": 3.7664287901861826e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.7686852465412812e-06, "reward_total_composite_mean": 0.9981523752212524, "reward_total_composite_std": 3.7664287901861826e-06, "reward_total_mean": 0.9981523752212524, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981523752212524, "rewards/meter/std": 3.7664287901861826e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981523752212524, "rewards/total_composite/std": 3.7664287901861826e-06, "sampling/importance_sampling_ratio/max": 1.4789509773254395, "sampling/importance_sampling_ratio/mean": 1.0037543773651123, "sampling/importance_sampling_ratio/min": 0.8453654646873474, "sampling/sampling_logp_difference/max": 0.39133310317993164, "sampling/sampling_logp_difference/mean": 0.006302142050117254, "step": 3057 }, { "clip_ratio/high_max": 0.0038170163752511144, "clip_ratio/high_mean": 0.0038170163752511144, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.005710955825634301, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.75, "completions/mean_terminated_length": 65.75, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.07432558294385672, "epoch": 0.12282604329838936, "frac_reward_zero_std": 0.0, "grad_norm": 2.2398717403411865, "learning_rate": 7.363636363636364e-07, "loss": 0.0021, "num_tokens": 6949869.0, "reward": 0.9914587736129761, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9914587736129761, "reward_meter_std": 0.00509608956053853, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005096097476780415, "reward_total_composite_mean": 0.9914587736129761, "reward_total_composite_std": 0.00509608956053853, "reward_total_mean": 0.9914587736129761, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9914587736129761, "rewards/meter/std": 0.00509608956053853, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9914587736129761, "rewards/total_composite/std": 0.00509608956053853, "sampling/importance_sampling_ratio/max": 1.2646030187606812, "sampling/importance_sampling_ratio/mean": 1.003832221031189, "sampling/importance_sampling_ratio/min": 0.42544955015182495, "sampling/sampling_logp_difference/max": 0.8546088933944702, "sampling/sampling_logp_difference/mean": 0.011036393232643604, "step": 3058 }, { "clip_ratio/high_max": 0.006889763753861189, "clip_ratio/high_mean": 0.006889763753861189, "clip_ratio/low_mean": 0.008024971117265522, "clip_ratio/low_min": 0.008024971117265522, "clip_ratio/region_mean": 0.014914734871126711, "completions/clipped_ratio": 0.0, "completions/max_length": 127.0, "completions/max_terminated_length": 127.0, "completions/mean_length": 125.875, "completions/mean_terminated_length": 125.875, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.24840066954493523, "epoch": 0.12286620878017432, "frac_reward_zero_std": 0.0, "grad_norm": 2.9161179065704346, "learning_rate": 7.333333333333334e-07, "loss": -0.0012, "num_tokens": 6952388.0, "reward": 0.8751063346862793, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.938283383846283, "reward_meter_std": 0.15279191732406616, "reward_repeat_penalty_mean": 0.9285714626312256, "reward_repeat_penalty_std": 0.07636035233736038, "reward_std": 0.17424653470516205, "reward_total_composite_mean": 0.8751063346862793, "reward_total_composite_std": 0.17424653470516205, "reward_total_mean": 0.8751063346862793, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.938283383846283, "rewards/meter/std": 0.15279191732406616, "rewards/repeat_penalty/mean": 0.9285714626312256, "rewards/repeat_penalty/std": 0.07636035233736038, "rewards/total_composite/mean": 0.8751063346862793, "rewards/total_composite/std": 0.17424653470516205, "sampling/importance_sampling_ratio/max": 1.7322660684585571, "sampling/importance_sampling_ratio/mean": 1.0073537826538086, "sampling/importance_sampling_ratio/min": 0.3819577693939209, "sampling/sampling_logp_difference/max": 0.9624452590942383, "sampling/sampling_logp_difference/mean": 0.025627773255109787, "step": 3059 }, { "clip_ratio/high_max": 0.01723209215560928, "clip_ratio/high_mean": 0.01723209215560928, "clip_ratio/low_mean": 0.010748314904049039, "clip_ratio/low_min": 0.010748314904049039, "clip_ratio/region_mean": 0.02798040705965832, "completions/clipped_ratio": 0.0, "completions/max_length": 145.0, "completions/max_terminated_length": 145.0, "completions/mean_length": 138.125, "completions/mean_terminated_length": 138.125, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.32595139369368553, "epoch": 0.12290637426195927, "frac_reward_zero_std": 0.0, "grad_norm": 3.135732650756836, "learning_rate": 7.303030303030304e-07, "loss": 0.0128, "num_tokens": 6954749.0, "reward": 0.9924278259277344, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9924278259277344, "reward_meter_std": 0.0035289754159748554, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0035289691295474768, "reward_total_composite_mean": 0.9924278259277344, "reward_total_composite_std": 0.0035289754159748554, "reward_total_mean": 0.9924278259277344, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9924278259277344, "rewards/meter/std": 0.0035289754159748554, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924278259277344, "rewards/total_composite/std": 0.0035289754159748554, "sampling/importance_sampling_ratio/max": 1.7354730367660522, "sampling/importance_sampling_ratio/mean": 1.0079230070114136, "sampling/importance_sampling_ratio/min": 0.37118610739707947, "sampling/sampling_logp_difference/max": 0.9910516738891602, "sampling/sampling_logp_difference/mean": 0.03332981467247009, "step": 3060 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.000393166919820942, "epoch": 0.12294653974374423, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.272727272727273e-07, "loss": 0.0, "num_tokens": 6956349.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.000898838043213, "sampling/importance_sampling_ratio/mean": 1.0000483989715576, "sampling/importance_sampling_ratio/min": 0.9998447299003601, "sampling/sampling_logp_difference/max": 0.0008984014857560396, "sampling/sampling_logp_difference/mean": 5.019194941269234e-05, "step": 3061 }, { "clip_ratio/high_max": 0.01407297421246767, "clip_ratio/high_mean": 0.01407297421246767, "clip_ratio/low_mean": 0.0010000000474974513, "clip_ratio/low_min": 0.0010000000474974513, "clip_ratio/region_mean": 0.015072974259965122, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 124.625, "completions/mean_terminated_length": 124.625, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.1536051593720913, "epoch": 0.12298670522552918, "frac_reward_zero_std": 0.0, "grad_norm": 1.910351037979126, "learning_rate": 7.242424242424243e-07, "loss": 0.0008, "num_tokens": 6958746.0, "reward": 0.997795820236206, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997795820236206, "reward_meter_std": 0.0004729351494461298, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004729465872514993, "reward_total_composite_mean": 0.997795820236206, "reward_total_composite_std": 0.0004729351494461298, "reward_total_mean": 0.997795820236206, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997795820236206, "rewards/meter/std": 0.0004729351494461298, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997795820236206, "rewards/total_composite/std": 0.0004729351494461298, "sampling/importance_sampling_ratio/max": 1.4670944213867188, "sampling/importance_sampling_ratio/mean": 1.0041706562042236, "sampling/importance_sampling_ratio/min": 0.13802744448184967, "sampling/sampling_logp_difference/max": 1.9803028106689453, "sampling/sampling_logp_difference/mean": 0.018115131184458733, "step": 3062 }, { "clip_ratio/high_max": 0.018338565016165376, "clip_ratio/high_mean": 0.018338565016165376, "clip_ratio/low_mean": 0.014415563317015767, "clip_ratio/low_min": 0.014415563317015767, "clip_ratio/region_mean": 0.03275412833318114, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 137.625, "completions/mean_terminated_length": 137.625, "completions/min_length": 134.0, "completions/min_terminated_length": 134.0, "entropy": 0.3563910871744156, "epoch": 0.12302687070731413, "frac_reward_zero_std": 0.0, "grad_norm": 3.4471962451934814, "learning_rate": 7.212121212121213e-07, "loss": 0.0173, "num_tokens": 6961175.0, "reward": 0.9925941228866577, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9925941228866577, "reward_meter_std": 0.0024486577603965998, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0024486577603965998, "reward_total_composite_mean": 0.9925941228866577, "reward_total_composite_std": 0.0024486577603965998, "reward_total_mean": 0.9925941228866577, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9925941228866577, "rewards/meter/std": 0.0024486577603965998, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9925941228866577, "rewards/total_composite/std": 0.0024486577603965998, "sampling/importance_sampling_ratio/max": 1.8253576755523682, "sampling/importance_sampling_ratio/mean": 1.0079936981201172, "sampling/importance_sampling_ratio/min": 0.2668985426425934, "sampling/sampling_logp_difference/max": 1.3208866119384766, "sampling/sampling_logp_difference/mean": 0.04126367345452309, "step": 3063 }, { "clip_ratio/high_max": 0.009527972200885415, "clip_ratio/high_mean": 0.009527972200885415, "clip_ratio/low_mean": 0.009565269108861685, "clip_ratio/low_min": 0.009565269108861685, "clip_ratio/region_mean": 0.0190932413097471, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.25, "completions/mean_terminated_length": 65.25, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.20045749936252832, "epoch": 0.12306703618909909, "frac_reward_zero_std": 0.0, "grad_norm": 3.6814091205596924, "learning_rate": 7.181818181818182e-07, "loss": 0.0097, "num_tokens": 6963041.0, "reward": 0.9920260310173035, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9920260310173035, "reward_meter_std": 0.0032561845146119595, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0032561977859586477, "reward_total_composite_mean": 0.9920260310173035, "reward_total_composite_std": 0.0032561845146119595, "reward_total_mean": 0.9920260310173035, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9920260310173035, "rewards/meter/std": 0.0032561845146119595, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9920260310173035, "rewards/total_composite/std": 0.0032561845146119595, "sampling/importance_sampling_ratio/max": 1.6320992708206177, "sampling/importance_sampling_ratio/mean": 1.0053472518920898, "sampling/importance_sampling_ratio/min": 0.2513670027256012, "sampling/sampling_logp_difference/max": 1.3808412551879883, "sampling/sampling_logp_difference/mean": 0.031295765191316605, "step": 3064 }, { "clip_ratio/high_max": 0.032739987364038825, "clip_ratio/high_mean": 0.032739987364038825, "clip_ratio/low_mean": 0.004285714123398066, "clip_ratio/low_min": 0.004285714123398066, "clip_ratio/region_mean": 0.03702570148743689, "completions/clipped_ratio": 0.0, "completions/max_length": 362.0, "completions/max_terminated_length": 362.0, "completions/mean_length": 350.25, "completions/mean_terminated_length": 350.25, "completions/min_length": 343.0, "completions/min_terminated_length": 343.0, "entropy": 0.4838559664785862, "epoch": 0.12310720167088404, "frac_reward_zero_std": 0.0, "grad_norm": 1.8719370365142822, "learning_rate": 7.151515151515153e-07, "loss": 0.0003, "num_tokens": 6967523.0, "reward": 0.8961663246154785, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957402944564819, "reward_meter_std": 0.009437782689929008, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.008493990637362003, "reward_total_composite_mean": 0.8961663246154785, "reward_total_composite_std": 0.008493990637362003, "reward_total_mean": 0.8961663246154785, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957402944564819, "rewards/meter/std": 0.009437782689929008, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.8961663246154785, "rewards/total_composite/std": 0.008493990637362003, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0094969272613525, "sampling/importance_sampling_ratio/min": 0.12798814475536346, "sampling/sampling_logp_difference/max": 2.0558176040649414, "sampling/sampling_logp_difference/mean": 0.05361352488398552, "step": 3065 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 36.0, "completions/mean_terminated_length": 36.0, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.013131682993844151, "epoch": 0.123147367152669, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.121212121212122e-07, "loss": 0.0, "num_tokens": 6969011.0, "reward": 0.9996045231819153, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9996045231819153, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0258433818817139, "sampling/importance_sampling_ratio/mean": 1.0010969638824463, "sampling/importance_sampling_ratio/min": 0.9661255478858948, "sampling/sampling_logp_difference/max": 0.034461501985788345, "sampling/sampling_logp_difference/mean": 0.0014014473417773843, "step": 3066 }, { "clip_ratio/high_max": 0.010746001382358372, "clip_ratio/high_mean": 0.010746001382358372, "clip_ratio/low_mean": 0.0036498295376077294, "clip_ratio/low_min": 0.0036498295376077294, "clip_ratio/region_mean": 0.014395830919966102, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 69.25, "completions/mean_terminated_length": 69.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.21143894083797932, "epoch": 0.12318753263445395, "frac_reward_zero_std": 0.0, "grad_norm": 3.1952340602874756, "learning_rate": 7.090909090909092e-07, "loss": -0.0038, "num_tokens": 6970813.0, "reward": 0.9955199956893921, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9955199956893921, "reward_meter_std": 0.0008645570487715304, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0008645570487715304, "reward_total_composite_mean": 0.9955199956893921, "reward_total_composite_std": 0.0008645570487715304, "reward_total_mean": 0.9955199956893921, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9955199956893921, "rewards/meter/std": 0.0008645570487715304, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9955199956893921, "rewards/total_composite/std": 0.0008645570487715304, "sampling/importance_sampling_ratio/max": 1.5354810953140259, "sampling/importance_sampling_ratio/mean": 1.0018832683563232, "sampling/importance_sampling_ratio/min": 0.29611852765083313, "sampling/sampling_logp_difference/max": 1.2169954776763916, "sampling/sampling_logp_difference/mean": 0.02916860766708851, "step": 3067 }, { "clip_ratio/high_max": 0.014811165689025074, "clip_ratio/high_mean": 0.014811165689025074, "clip_ratio/low_mean": 0.0007022471982054412, "clip_ratio/low_min": 0.0007022471982054412, "clip_ratio/region_mean": 0.015513412887230515, "completions/clipped_ratio": 0.0, "completions/max_length": 179.0, "completions/max_terminated_length": 179.0, "completions/mean_length": 177.375, "completions/mean_terminated_length": 177.375, "completions/min_length": 176.0, "completions/min_terminated_length": 176.0, "entropy": 0.22719617374241352, "epoch": 0.1232276981162389, "frac_reward_zero_std": 0.0, "grad_norm": 1.5282530784606934, "learning_rate": 7.060606060606061e-07, "loss": 0.0041, "num_tokens": 6973664.0, "reward": 0.9852460026741028, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991249442100525, "reward_meter_std": 0.00017947136075235903, "reward_repeat_penalty_mean": 0.9861111044883728, "reward_repeat_penalty_std": 0.03928370773792267, "reward_std": 0.03919193148612976, "reward_total_composite_mean": 0.9852460026741028, "reward_total_composite_std": 0.039191942662000656, "reward_total_mean": 0.9852460026741028, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991249442100525, "rewards/meter/std": 0.00017947136075235903, "rewards/repeat_penalty/mean": 0.9861111044883728, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.9852460026741028, "rewards/total_composite/std": 0.039191942662000656, "sampling/importance_sampling_ratio/max": 1.8010437488555908, "sampling/importance_sampling_ratio/mean": 1.003198266029358, "sampling/importance_sampling_ratio/min": 0.33639028668403625, "sampling/sampling_logp_difference/max": 1.0894832611083984, "sampling/sampling_logp_difference/mean": 0.02057509496808052, "step": 3068 }, { "clip_ratio/high_max": 0.007437994237989187, "clip_ratio/high_mean": 0.007437994237989187, "clip_ratio/low_mean": 0.005277339863823727, "clip_ratio/low_min": 0.005277339863823727, "clip_ratio/region_mean": 0.012715334101812914, "completions/clipped_ratio": 0.0, "completions/max_length": 288.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 285.375, "completions/mean_terminated_length": 285.375, "completions/min_length": 282.0, "completions/min_terminated_length": 282.0, "entropy": 0.3014299161732197, "epoch": 0.12326786359802386, "frac_reward_zero_std": 0.0, "grad_norm": 1.41837477684021, "learning_rate": 7.03030303030303e-07, "loss": 0.0046, "num_tokens": 6977475.0, "reward": 0.9490443468093872, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989954233169556, "reward_meter_std": 0.00014736292359884828, "reward_repeat_penalty_mean": 0.9500000476837158, "reward_repeat_penalty_std": 0.0471404492855072, "reward_std": 0.04706757515668869, "reward_total_composite_mean": 0.9490443468093872, "reward_total_composite_std": 0.04706759378314018, "reward_total_mean": 0.9490443468093872, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989954233169556, "rewards/meter/std": 0.00014736292359884828, "rewards/repeat_penalty/mean": 0.9500000476837158, "rewards/repeat_penalty/std": 0.0471404492855072, "rewards/total_composite/mean": 0.9490443468093872, "rewards/total_composite/std": 0.04706759378314018, "sampling/importance_sampling_ratio/max": 1.655118703842163, "sampling/importance_sampling_ratio/mean": 1.005640983581543, "sampling/importance_sampling_ratio/min": 0.2844875454902649, "sampling/sampling_logp_difference/max": 1.257065773010254, "sampling/sampling_logp_difference/mean": 0.026732422411441803, "step": 3069 }, { "clip_ratio/high_max": 0.031633416656404734, "clip_ratio/high_mean": 0.031633416656404734, "clip_ratio/low_mean": 0.005974947940558195, "clip_ratio/low_min": 0.005974947940558195, "clip_ratio/region_mean": 0.03760836459696293, "completions/clipped_ratio": 0.0, "completions/max_length": 250.0, "completions/max_terminated_length": 250.0, "completions/mean_length": 234.25, "completions/mean_terminated_length": 234.25, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.43540840223431587, "epoch": 0.12330802907980881, "frac_reward_zero_std": 0.0, "grad_norm": 3.581768751144409, "learning_rate": 7.000000000000001e-07, "loss": 0.0073, "num_tokens": 6981013.0, "reward": 0.9766829013824463, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9958624839782715, "reward_meter_std": 0.004256864078342915, "reward_repeat_penalty_mean": 0.9807692170143127, "reward_repeat_penalty_std": 0.03560846298933029, "reward_std": 0.03484949469566345, "reward_total_composite_mean": 0.9766829013824463, "reward_total_composite_std": 0.034849490970373154, "reward_total_mean": 0.9766829013824463, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9958624839782715, "rewards/meter/std": 0.004256864078342915, "rewards/repeat_penalty/mean": 0.9807692170143127, "rewards/repeat_penalty/std": 0.03560846298933029, "rewards/total_composite/mean": 0.9766829013824463, "rewards/total_composite/std": 0.034849490970373154, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0083545446395874, "sampling/importance_sampling_ratio/min": 0.026942478492856026, "sampling/sampling_logp_difference/max": 3.614051103591919, "sampling/sampling_logp_difference/mean": 0.05148389935493469, "step": 3070 }, { "clip_ratio/high_max": 0.0025641699321568012, "clip_ratio/high_mean": 0.0025641699321568012, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/region_mean": 0.005115190288051963, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.875, "completions/mean_terminated_length": 97.875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.02978663402609527, "epoch": 0.12334819456159377, "frac_reward_zero_std": 0.0, "grad_norm": 0.05822064355015755, "learning_rate": 6.969696969696971e-07, "loss": 0.0005, "num_tokens": 6982996.0, "reward": 0.9994089603424072, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994089603424072, "reward_meter_std": 7.70060796639882e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.685384844080545e-06, "reward_total_composite_mean": 0.9994089603424072, "reward_total_composite_std": 7.70060796639882e-06, "reward_total_mean": 0.9994089603424072, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994089603424072, "rewards/meter/std": 7.70060796639882e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994089603424072, "rewards/total_composite/std": 7.70060796639882e-06, "sampling/importance_sampling_ratio/max": 1.2397805452346802, "sampling/importance_sampling_ratio/mean": 1.0012834072113037, "sampling/importance_sampling_ratio/min": 0.5212281346321106, "sampling/sampling_logp_difference/max": 0.6515674591064453, "sampling/sampling_logp_difference/mean": 0.004502264782786369, "step": 3071 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0010117705096490681, "epoch": 0.12338836004337872, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.939393939393941e-07, "loss": 0.0, "num_tokens": 6984748.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0020101070404053, "sampling/importance_sampling_ratio/mean": 1.000089168548584, "sampling/importance_sampling_ratio/min": 0.9984797239303589, "sampling/sampling_logp_difference/max": 0.002008092822507024, "sampling/sampling_logp_difference/mean": 0.00010368959920015186, "step": 3072 }, { "clip_ratio/high_max": 0.007463517948053777, "clip_ratio/high_mean": 0.007463517948053777, "clip_ratio/low_mean": 0.0018939394503831863, "clip_ratio/low_min": 0.0018939394503831863, "clip_ratio/region_mean": 0.009357457398436964, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.056920765433460474, "epoch": 0.12342852552516367, "frac_reward_zero_std": 0.0, "grad_norm": 0.6883857846260071, "learning_rate": 6.90909090909091e-07, "loss": 0.0005, "num_tokens": 6986610.0, "reward": 0.9981359243392944, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981359243392944, "reward_meter_std": 3.569047839846462e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.56822092726361e-05, "reward_total_composite_mean": 0.9981359243392944, "reward_total_composite_std": 3.569047839846462e-05, "reward_total_mean": 0.9981359243392944, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981359243392944, "rewards/meter/std": 3.569047839846462e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981359243392944, "rewards/total_composite/std": 3.569047839846462e-05, "sampling/importance_sampling_ratio/max": 1.4870829582214355, "sampling/importance_sampling_ratio/mean": 0.9993820190429688, "sampling/importance_sampling_ratio/min": 0.3307732045650482, "sampling/sampling_logp_difference/max": 1.1063222885131836, "sampling/sampling_logp_difference/mean": 0.013685759156942368, "step": 3073 }, { "clip_ratio/high_max": 0.009702706767711788, "clip_ratio/high_mean": 0.009702706767711788, "clip_ratio/low_mean": 0.003496503457427025, "clip_ratio/low_min": 0.003496503457427025, "clip_ratio/region_mean": 0.013199210225138813, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 142.625, "completions/mean_terminated_length": 142.625, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.18659264594316483, "epoch": 0.12346869100694863, "frac_reward_zero_std": 0.0, "grad_norm": 1.0535986423492432, "learning_rate": 6.878787878787879e-07, "loss": 0.0032, "num_tokens": 6989111.0, "reward": 0.9813013076782227, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991422891616821, "reward_meter_std": 0.0001949365541804582, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.050480738282203674, "reward_total_composite_mean": 0.9813013076782227, "reward_total_composite_std": 0.05048074945807457, "reward_total_mean": 0.9813013076782227, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991422891616821, "rewards/meter/std": 0.0001949365541804582, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9813013076782227, "rewards/total_composite/std": 0.05048074945807457, "sampling/importance_sampling_ratio/max": 1.7058764696121216, "sampling/importance_sampling_ratio/mean": 1.0043222904205322, "sampling/importance_sampling_ratio/min": 0.18417169153690338, "sampling/sampling_logp_difference/max": 1.6918869018554688, "sampling/sampling_logp_difference/mean": 0.018873082473874092, "step": 3074 }, { "clip_ratio/high_max": 0.021908456226810813, "clip_ratio/high_mean": 0.021908456226810813, "clip_ratio/low_mean": 0.0034246575087308884, "clip_ratio/low_min": 0.0034246575087308884, "clip_ratio/region_mean": 0.0253331137355417, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 69.25, "completions/mean_terminated_length": 69.25, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.24566587805747986, "epoch": 0.12350885648873358, "frac_reward_zero_std": 0.0, "grad_norm": 3.619894504547119, "learning_rate": 6.848484848484849e-07, "loss": 0.0244, "num_tokens": 6990969.0, "reward": 0.9823685884475708, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9823685884475708, "reward_meter_std": 0.03472607955336571, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.034726083278656006, "reward_total_composite_mean": 0.9823685884475708, "reward_total_composite_std": 0.03472607955336571, "reward_total_mean": 0.9823685884475708, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9823685884475708, "rewards/meter/std": 0.03472607955336571, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9823685884475708, "rewards/total_composite/std": 0.03472607955336571, "sampling/importance_sampling_ratio/max": 1.8515444993972778, "sampling/importance_sampling_ratio/mean": 1.006119728088379, "sampling/importance_sampling_ratio/min": 0.24358773231506348, "sampling/sampling_logp_difference/max": 1.412278175354004, "sampling/sampling_logp_difference/mean": 0.030778199434280396, "step": 3075 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/region_mean": 0.004098360426723957, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.02187307458370924, "epoch": 0.12354902197051854, "frac_reward_zero_std": 0.0, "grad_norm": 0.014152090065181255, "learning_rate": 6.818181818181818e-07, "loss": 0.0003, "num_tokens": 6992713.0, "reward": 0.997339129447937, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997339129447937, "reward_meter_std": 3.875939285080676e-07, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.8628223819614504e-07, "reward_total_composite_mean": 0.997339129447937, "reward_total_composite_std": 3.875939285080676e-07, "reward_total_mean": 0.997339129447937, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997339129447937, "rewards/meter/std": 3.875939285080676e-07, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997339129447937, "rewards/total_composite/std": 3.875939285080676e-07, "sampling/importance_sampling_ratio/max": 1.2609317302703857, "sampling/importance_sampling_ratio/mean": 0.9999513030052185, "sampling/importance_sampling_ratio/min": 0.5995131134986877, "sampling/sampling_logp_difference/max": 0.5116375088691711, "sampling/sampling_logp_difference/mean": 0.004271084442734718, "step": 3076 }, { "clip_ratio/high_max": 0.0018115942366421223, "clip_ratio/high_mean": 0.0018115942366421223, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.0036498295376077294, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.050312931183725595, "epoch": 0.12358918745230349, "frac_reward_zero_std": 0.0, "grad_norm": 0.27777397632598877, "learning_rate": 6.78787878787879e-07, "loss": 0.0002, "num_tokens": 6994530.0, "reward": 0.9994935989379883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994935989379883, "reward_meter_std": 8.932632226787973e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.932575838116463e-06, "reward_total_composite_mean": 0.9994935989379883, "reward_total_composite_std": 8.932632226787973e-06, "reward_total_mean": 0.9994935989379883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994935989379883, "rewards/meter/std": 8.932632226787973e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994935989379883, "rewards/total_composite/std": 8.932632226787973e-06, "sampling/importance_sampling_ratio/max": 1.1400055885314941, "sampling/importance_sampling_ratio/mean": 1.0012462139129639, "sampling/importance_sampling_ratio/min": 0.28703317046165466, "sampling/sampling_logp_difference/max": 1.2481575012207031, "sampling/sampling_logp_difference/mean": 0.00869977567344904, "step": 3077 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.032520961947739124, "epoch": 0.12362935293408844, "frac_reward_zero_std": 0.0, "grad_norm": 0.11041238158941269, "learning_rate": 6.757575757575759e-07, "loss": -0.0002, "num_tokens": 6996666.0, "reward": 0.9994069337844849, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994069337844849, "reward_meter_std": 4.966230790159898e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.970342160959262e-06, "reward_total_composite_mean": 0.9994069337844849, "reward_total_composite_std": 4.966230790159898e-06, "reward_total_mean": 0.9994069337844849, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994069337844849, "rewards/meter/std": 4.966230790159898e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994069337844849, "rewards/total_composite/std": 4.966230790159898e-06, "sampling/importance_sampling_ratio/max": 1.217292070388794, "sampling/importance_sampling_ratio/mean": 0.9998006820678711, "sampling/importance_sampling_ratio/min": 0.5330917835235596, "sampling/sampling_logp_difference/max": 0.6290616989135742, "sampling/sampling_logp_difference/mean": 0.003691497491672635, "step": 3078 }, { "clip_ratio/high_max": 0.009800075204111636, "clip_ratio/high_mean": 0.009800075204111636, "clip_ratio/low_mean": 0.004197834758087993, "clip_ratio/low_min": 0.004197834758087993, "clip_ratio/region_mean": 0.013997909962199628, "completions/clipped_ratio": 0.0, "completions/max_length": 179.0, "completions/max_terminated_length": 179.0, "completions/mean_length": 178.375, "completions/mean_terminated_length": 178.375, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.23156088776886463, "epoch": 0.1236695184158734, "frac_reward_zero_std": 0.0, "grad_norm": 1.676236867904663, "learning_rate": 6.727272727272728e-07, "loss": 0.0004, "num_tokens": 6999685.0, "reward": 0.9575244188308716, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991593360900879, "reward_meter_std": 8.344750676769763e-05, "reward_repeat_penalty_mean": 0.9583333134651184, "reward_repeat_penalty_std": 0.08266931027173996, "reward_std": 0.08255956321954727, "reward_total_composite_mean": 0.9575244188308716, "reward_total_composite_std": 0.08255957812070847, "reward_total_mean": 0.9575244188308716, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991593360900879, "rewards/meter/std": 8.344750676769763e-05, "rewards/repeat_penalty/mean": 0.9583333134651184, "rewards/repeat_penalty/std": 0.08266931027173996, "rewards/total_composite/mean": 0.9575244188308716, "rewards/total_composite/std": 0.08255957812070847, "sampling/importance_sampling_ratio/max": 1.5720078945159912, "sampling/importance_sampling_ratio/mean": 1.0064208507537842, "sampling/importance_sampling_ratio/min": 0.3305152654647827, "sampling/sampling_logp_difference/max": 1.107102394104004, "sampling/sampling_logp_difference/mean": 0.020297784358263016, "step": 3079 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 79.875, "completions/mean_terminated_length": 79.875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.0046348033065442, "epoch": 0.12370968389765835, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.696969696969698e-07, "loss": 0.0, "num_tokens": 7001620.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0144623517990112, "sampling/importance_sampling_ratio/mean": 1.0004509687423706, "sampling/importance_sampling_ratio/min": 0.9983435869216919, "sampling/sampling_logp_difference/max": 0.014358794316649437, "sampling/sampling_logp_difference/mean": 0.0004583710106089711, "step": 3080 }, { "clip_ratio/high_max": 0.02169638266786933, "clip_ratio/high_mean": 0.02169638266786933, "clip_ratio/low_mean": 0.005876097013242543, "clip_ratio/low_min": 0.005876097013242543, "clip_ratio/region_mean": 0.027572479681111872, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 126.125, "completions/mean_terminated_length": 126.125, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.36210910603404045, "epoch": 0.1237498493794433, "frac_reward_zero_std": 0.0, "grad_norm": 2.4082579612731934, "learning_rate": 6.666666666666667e-07, "loss": 0.0079, "num_tokens": 7004021.0, "reward": 0.9353639483451843, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9883460402488708, "reward_meter_std": 0.006720090284943581, "reward_repeat_penalty_mean": 0.9464285969734192, "reward_repeat_penalty_std": 0.07393559068441391, "reward_std": 0.07287097722291946, "reward_total_composite_mean": 0.9353639483451843, "reward_total_composite_std": 0.07287098467350006, "reward_total_mean": 0.9353639483451843, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9883460402488708, "rewards/meter/std": 0.006720090284943581, "rewards/repeat_penalty/mean": 0.9464285969734192, "rewards/repeat_penalty/std": 0.07393559068441391, "rewards/total_composite/mean": 0.9353639483451843, "rewards/total_composite/std": 0.07287098467350006, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0118496417999268, "sampling/importance_sampling_ratio/min": 0.243983656167984, "sampling/sampling_logp_difference/max": 1.410654067993164, "sampling/sampling_logp_difference/mean": 0.03183456510305405, "step": 3081 }, { "clip_ratio/high_max": 0.03181355516426265, "clip_ratio/high_mean": 0.03181355516426265, "clip_ratio/low_mean": 0.011420327704399824, "clip_ratio/low_min": 0.011420327704399824, "clip_ratio/region_mean": 0.04323388286866248, "completions/clipped_ratio": 0.0, "completions/max_length": 340.0, "completions/max_terminated_length": 340.0, "completions/mean_length": 331.875, "completions/mean_terminated_length": 331.875, "completions/min_length": 316.0, "completions/min_terminated_length": 316.0, "entropy": 0.594636969268322, "epoch": 0.12379001486122826, "frac_reward_zero_std": 0.0, "grad_norm": 3.0218822956085205, "learning_rate": 6.636363636363636e-07, "loss": 0.0066, "num_tokens": 7008316.0, "reward": 0.8187316656112671, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9956144094467163, "reward_meter_std": 0.0046219234354794025, "reward_repeat_penalty_mean": 0.9868420958518982, "reward_repeat_penalty_std": 0.024363677948713303, "reward_std": 0.019191952422261238, "reward_total_composite_mean": 0.8187316656112671, "reward_total_composite_std": 0.019191961735486984, "reward_total_mean": 0.8187316656112671, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9956144094467163, "rewards/meter/std": 0.0046219234354794025, "rewards/repeat_penalty/mean": 0.9868420958518982, "rewards/repeat_penalty/std": 0.024363677948713303, "rewards/total_composite/mean": 0.8187316656112671, "rewards/total_composite/std": 0.019191961735486984, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.016357183456421, "sampling/importance_sampling_ratio/min": 0.018101176247000694, "sampling/sampling_logp_difference/max": 4.011778354644775, "sampling/sampling_logp_difference/mean": 0.06327174603939056, "step": 3082 }, { "clip_ratio/high_max": 0.023763853823766112, "clip_ratio/high_mean": 0.023763853823766112, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.02560208912473172, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.375, "completions/mean_terminated_length": 68.375, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.12113892659544945, "epoch": 0.12383018034301321, "frac_reward_zero_std": 0.0, "grad_norm": 3.886202335357666, "learning_rate": 6.606060606060606e-07, "loss": -0.001, "num_tokens": 7010295.0, "reward": 0.9993771314620972, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993771314620972, "reward_meter_std": 0.00016229395987465978, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00016229994071181864, "reward_total_composite_mean": 0.9993771314620972, "reward_total_composite_std": 0.00016229395987465978, "reward_total_mean": 0.9993771314620972, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993771314620972, "rewards/meter/std": 0.00016229395987465978, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993771314620972, "rewards/total_composite/std": 0.00016229395987465978, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.9933637380599976, "sampling/importance_sampling_ratio/min": 0.03502493351697922, "sampling/sampling_logp_difference/max": 3.3516950607299805, "sampling/sampling_logp_difference/mean": 0.0466373972594738, "step": 3083 }, { "clip_ratio/high_max": 0.03329520847182721, "clip_ratio/high_mean": 0.03329520847182721, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/region_mean": 0.036918396945111454, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 70.75, "completions/mean_terminated_length": 70.75, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2923265937715769, "epoch": 0.12387034582479817, "frac_reward_zero_std": 0.0, "grad_norm": 7.438833713531494, "learning_rate": 6.575757575757575e-07, "loss": -0.0038, "num_tokens": 7012093.0, "reward": 0.9911935925483704, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9911935925483704, "reward_meter_std": 0.011583052575588226, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.011583039537072182, "reward_total_composite_mean": 0.9911935925483704, "reward_total_composite_std": 0.011583052575588226, "reward_total_mean": 0.9911935925483704, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9911935925483704, "rewards/meter/std": 0.011583052575588226, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9911935925483704, "rewards/total_composite/std": 0.011583052575588226, "sampling/importance_sampling_ratio/max": 1.5489575862884521, "sampling/importance_sampling_ratio/mean": 0.999201238155365, "sampling/importance_sampling_ratio/min": 0.3502725660800934, "sampling/sampling_logp_difference/max": 1.0490436553955078, "sampling/sampling_logp_difference/mean": 0.034917399287223816, "step": 3084 }, { "clip_ratio/high_max": 0.01920302864164114, "clip_ratio/high_mean": 0.01920302864164114, "clip_ratio/low_mean": 0.006369811948388815, "clip_ratio/low_min": 0.006369811948388815, "clip_ratio/region_mean": 0.025572840590029955, "completions/clipped_ratio": 0.0, "completions/max_length": 159.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 156.25, "completions/mean_terminated_length": 156.25, "completions/min_length": 150.0, "completions/min_terminated_length": 150.0, "entropy": 0.3303154893219471, "epoch": 0.12391051130658312, "frac_reward_zero_std": 0.0, "grad_norm": 1.1208101511001587, "learning_rate": 6.545454545454547e-07, "loss": 0.0045, "num_tokens": 7014791.0, "reward": 0.9991309642791748, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991309642791748, "reward_meter_std": 0.00010435016884002835, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010435186413815245, "reward_total_composite_mean": 0.9991309642791748, "reward_total_composite_std": 0.00010435016884002835, "reward_total_mean": 0.9991309642791748, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991309642791748, "rewards/meter/std": 0.00010435016884002835, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991309642791748, "rewards/total_composite/std": 0.00010435016884002835, "sampling/importance_sampling_ratio/max": 1.801207184791565, "sampling/importance_sampling_ratio/mean": 1.0100115537643433, "sampling/importance_sampling_ratio/min": 0.456314355134964, "sampling/sampling_logp_difference/max": 0.7845733165740967, "sampling/sampling_logp_difference/mean": 0.03277255594730377, "step": 3085 }, { "clip_ratio/high_max": 0.028133108280599117, "clip_ratio/high_mean": 0.028133108280599117, "clip_ratio/low_mean": 0.010255418019369245, "clip_ratio/low_min": 0.010255418019369245, "clip_ratio/region_mean": 0.03838852629996836, "completions/clipped_ratio": 0.0, "completions/max_length": 136.0, "completions/max_terminated_length": 136.0, "completions/mean_length": 133.625, "completions/mean_terminated_length": 133.625, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.2722382918000221, "epoch": 0.12395067678836807, "frac_reward_zero_std": 0.0, "grad_norm": 4.402876377105713, "learning_rate": 6.515151515151516e-07, "loss": 0.0065, "num_tokens": 7017132.0, "reward": 0.998715341091156, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998715341091156, "reward_meter_std": 0.0006201790529303253, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00062018126482144, "reward_total_composite_mean": 0.998715341091156, "reward_total_composite_std": 0.0006201790529303253, "reward_total_mean": 0.998715341091156, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998715341091156, "rewards/meter/std": 0.0006201790529303253, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998715341091156, "rewards/total_composite/std": 0.0006201790529303253, "sampling/importance_sampling_ratio/max": 1.980418086051941, "sampling/importance_sampling_ratio/mean": 1.0022952556610107, "sampling/importance_sampling_ratio/min": 0.16260430216789246, "sampling/sampling_logp_difference/max": 1.8164355754852295, "sampling/sampling_logp_difference/mean": 0.03640494868159294, "step": 3086 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0020491802133619785, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.016273654997348785, "epoch": 0.12399084227015303, "frac_reward_zero_std": 0.0, "grad_norm": 0.165548175573349, "learning_rate": 6.484848484848485e-07, "loss": -0.0, "num_tokens": 7018852.0, "reward": 0.9973361492156982, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973361492156982, "reward_meter_std": 8.160675861290656e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.157768206729088e-06, "reward_total_composite_mean": 0.9973361492156982, "reward_total_composite_std": 8.160675861290656e-06, "reward_total_mean": 0.9973361492156982, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973361492156982, "rewards/meter/std": 8.160675861290656e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973361492156982, "rewards/total_composite/std": 8.160675861290656e-06, "sampling/importance_sampling_ratio/max": 1.119980812072754, "sampling/importance_sampling_ratio/mean": 0.9994197487831116, "sampling/importance_sampling_ratio/min": 0.394413560628891, "sampling/sampling_logp_difference/max": 0.9303553104400635, "sampling/sampling_logp_difference/mean": 0.003540643723681569, "step": 3087 }, { "clip_ratio/high_max": 0.014212732203304768, "clip_ratio/high_mean": 0.014212732203304768, "clip_ratio/low_mean": 0.0020000000949949026, "clip_ratio/low_min": 0.0020000000949949026, "clip_ratio/region_mean": 0.01621273229829967, "completions/clipped_ratio": 0.0, "completions/max_length": 125.0, "completions/max_terminated_length": 125.0, "completions/mean_length": 123.5, "completions/mean_terminated_length": 123.5, "completions/min_length": 121.0, "completions/min_terminated_length": 121.0, "entropy": 0.1684292033314705, "epoch": 0.12403100775193798, "frac_reward_zero_std": 0.0, "grad_norm": 5.205334186553955, "learning_rate": 6.454545454545455e-07, "loss": 0.0063, "num_tokens": 7021320.0, "reward": 0.9796685576438904, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9974862337112427, "reward_meter_std": 0.00045686395606026053, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.050275154411792755, "reward_total_composite_mean": 0.9796685576438904, "reward_total_composite_std": 0.05027516558766365, "reward_total_mean": 0.9796685576438904, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9974862337112427, "rewards/meter/std": 0.00045686395606026053, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9796685576438904, "rewards/total_composite/std": 0.05027516558766365, "sampling/importance_sampling_ratio/max": 1.8565744161605835, "sampling/importance_sampling_ratio/mean": 1.004548192024231, "sampling/importance_sampling_ratio/min": 0.25753557682037354, "sampling/sampling_logp_difference/max": 1.3565974235534668, "sampling/sampling_logp_difference/mean": 0.0200046356767416, "step": 3088 }, { "clip_ratio/high_max": 0.007637429982423782, "clip_ratio/high_mean": 0.007637429982423782, "clip_ratio/low_mean": 0.01749891194049269, "clip_ratio/low_min": 0.01749891194049269, "clip_ratio/region_mean": 0.025136341922916472, "completions/clipped_ratio": 0.0, "completions/max_length": 397.0, "completions/max_terminated_length": 397.0, "completions/mean_length": 360.125, "completions/mean_terminated_length": 360.125, "completions/min_length": 344.0, "completions/min_terminated_length": 344.0, "entropy": 0.4755103662610054, "epoch": 0.12407117323372294, "frac_reward_zero_std": 0.0, "grad_norm": 1.6393530368804932, "learning_rate": 6.424242424242424e-07, "loss": -0.0405, "num_tokens": 7025761.0, "reward": 0.8282498717308044, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8409091234207153, "reward_count_adherence_std": 0.04208274558186531, "reward_meter_mean": 0.9991971254348755, "reward_meter_std": 0.00017116445815190673, "reward_repeat_penalty_mean": 0.9860681295394897, "reward_repeat_penalty_std": 0.02584986388683319, "reward_std": 0.04060585796833038, "reward_total_composite_mean": 0.8282498717308044, "reward_total_composite_std": 0.040605854243040085, "reward_total_mean": 0.8282498717308044, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8409091234207153, "rewards/count_adherence/std": 0.04208274558186531, "rewards/meter/mean": 0.9991971254348755, "rewards/meter/std": 0.00017116445815190673, "rewards/repeat_penalty/mean": 0.9860681295394897, "rewards/repeat_penalty/std": 0.02584986388683319, "rewards/total_composite/mean": 0.8282498717308044, "rewards/total_composite/std": 0.040605854243040085, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0105684995651245, "sampling/importance_sampling_ratio/min": 0.2684142589569092, "sampling/sampling_logp_difference/max": 1.3152236938476562, "sampling/sampling_logp_difference/mean": 0.04586139693856239, "step": 3089 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 79.875, "completions/mean_terminated_length": 79.875, "completions/min_length": 79.0, "completions/min_terminated_length": 79.0, "entropy": 0.004133307666052133, "epoch": 0.12411133871550789, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.393939393939394e-07, "loss": 0.0, "num_tokens": 7027640.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.015737533569336, "sampling/importance_sampling_ratio/mean": 0.9997469782829285, "sampling/importance_sampling_ratio/min": 0.5356513261795044, "sampling/sampling_logp_difference/max": 0.6242718696594238, "sampling/sampling_logp_difference/mean": 0.001453942502848804, "step": 3090 }, { "clip_ratio/high_max": 0.007018499891273677, "clip_ratio/high_mean": 0.007018499891273677, "clip_ratio/low_mean": 0.0034722222480922937, "clip_ratio/low_min": 0.0034722222480922937, "clip_ratio/region_mean": 0.010490722139365971, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.5, "completions/mean_terminated_length": 71.5, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.09367726277559996, "epoch": 0.12415150419729284, "frac_reward_zero_std": 0.0, "grad_norm": 0.5760459899902344, "learning_rate": 6.363636363636364e-07, "loss": 0.0015, "num_tokens": 7029612.0, "reward": 0.9994015097618103, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994015097618103, "reward_meter_std": 4.198703754809685e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.200140756438486e-05, "reward_total_composite_mean": 0.9994015097618103, "reward_total_composite_std": 4.198703754809685e-05, "reward_total_mean": 0.9994015097618103, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994015097618103, "rewards/meter/std": 4.198703754809685e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994015097618103, "rewards/total_composite/std": 4.198703754809685e-05, "sampling/importance_sampling_ratio/max": 1.4178498983383179, "sampling/importance_sampling_ratio/mean": 1.0001534223556519, "sampling/importance_sampling_ratio/min": 0.4072599709033966, "sampling/sampling_logp_difference/max": 0.8983035087585449, "sampling/sampling_logp_difference/mean": 0.010645559057593346, "step": 3091 }, { "clip_ratio/high_max": 0.01596410130150616, "clip_ratio/high_mean": 0.01596410130150616, "clip_ratio/low_mean": 0.0032327587250620127, "clip_ratio/low_min": 0.0032327587250620127, "clip_ratio/region_mean": 0.019196860026568174, "completions/clipped_ratio": 0.0, "completions/max_length": 120.0, "completions/max_terminated_length": 120.0, "completions/mean_length": 117.375, "completions/mean_terminated_length": 117.375, "completions/min_length": 116.0, "completions/min_terminated_length": 116.0, "entropy": 0.2854463141411543, "epoch": 0.1241916696790778, "frac_reward_zero_std": 0.0, "grad_norm": 2.74900484085083, "learning_rate": 6.333333333333334e-07, "loss": -0.0027, "num_tokens": 7031807.0, "reward": 0.9737505912780762, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9986400604248047, "reward_meter_std": 0.0013286188477650285, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07163627445697784, "reward_total_composite_mean": 0.9737505912780762, "reward_total_composite_std": 0.07163625955581665, "reward_total_mean": 0.9737505912780762, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9986400604248047, "rewards/meter/std": 0.0013286188477650285, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9737505912780762, "rewards/total_composite/std": 0.07163625955581665, "sampling/importance_sampling_ratio/max": 1.3800021409988403, "sampling/importance_sampling_ratio/mean": 1.0021259784698486, "sampling/importance_sampling_ratio/min": 0.306382954120636, "sampling/sampling_logp_difference/max": 1.1829195022583008, "sampling/sampling_logp_difference/mean": 0.03260769695043564, "step": 3092 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/region_mean": 0.004098360426723957, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.020028789876960218, "epoch": 0.12423183516086275, "frac_reward_zero_std": 0.0, "grad_norm": 0.078668013215065, "learning_rate": 6.303030303030304e-07, "loss": 0.0001, "num_tokens": 7033567.0, "reward": 0.9973331689834595, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973331689834595, "reward_meter_std": 1.0622773515933659e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.0613573067530524e-05, "reward_total_composite_mean": 0.9973331689834595, "reward_total_composite_std": 1.0622773515933659e-05, "reward_total_mean": 0.9973331689834595, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973331689834595, "rewards/meter/std": 1.0622773515933659e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973331689834595, "rewards/total_composite/std": 1.0622773515933659e-05, "sampling/importance_sampling_ratio/max": 1.7602647542953491, "sampling/importance_sampling_ratio/mean": 1.0025733709335327, "sampling/importance_sampling_ratio/min": 0.7023022770881653, "sampling/sampling_logp_difference/max": 0.5654642581939697, "sampling/sampling_logp_difference/mean": 0.0038269164506345987, "step": 3093 }, { "clip_ratio/high_max": 0.010245901066809893, "clip_ratio/high_mean": 0.010245901066809893, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.012295081280171871, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.024718819418922067, "epoch": 0.1242720006426477, "frac_reward_zero_std": 0.0, "grad_norm": 0.07660355418920517, "learning_rate": 6.272727272727273e-07, "loss": 0.0003, "num_tokens": 7035327.0, "reward": 0.9973366260528564, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973366260528564, "reward_meter_std": 8.457681360596325e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.46226976136677e-06, "reward_total_composite_mean": 0.9973366260528564, "reward_total_composite_std": 8.457681360596325e-06, "reward_total_mean": 0.9973366260528564, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973366260528564, "rewards/meter/std": 8.457681360596325e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973366260528564, "rewards/total_composite/std": 8.457681360596325e-06, "sampling/importance_sampling_ratio/max": 1.83155357837677, "sampling/importance_sampling_ratio/mean": 0.9982660412788391, "sampling/importance_sampling_ratio/min": 0.3371179401874542, "sampling/sampling_logp_difference/max": 1.087322473526001, "sampling/sampling_logp_difference/mean": 0.008617924526333809, "step": 3094 }, { "clip_ratio/high_max": 0.0019085081876255572, "clip_ratio/high_mean": 0.0019085081876255572, "clip_ratio/low_mean": 0.0028625954291783273, "clip_ratio/low_min": 0.0028625954291783273, "clip_ratio/region_mean": 0.0047711036168038845, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 131.25, "completions/mean_terminated_length": 131.25, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.08324715122580528, "epoch": 0.12431216612443266, "frac_reward_zero_std": 0.0, "grad_norm": 0.8752101063728333, "learning_rate": 6.242424242424243e-07, "loss": 0.0003, "num_tokens": 7037833.0, "reward": 0.9993793964385986, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993793964385986, "reward_meter_std": 9.467204654356465e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.466917254030704e-05, "reward_total_composite_mean": 0.9993793964385986, "reward_total_composite_std": 9.467204654356465e-05, "reward_total_mean": 0.9993793964385986, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993793964385986, "rewards/meter/std": 9.467204654356465e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993793964385986, "rewards/total_composite/std": 9.467204654356465e-05, "sampling/importance_sampling_ratio/max": 1.359288215637207, "sampling/importance_sampling_ratio/mean": 1.0029832124710083, "sampling/importance_sampling_ratio/min": 0.44786468148231506, "sampling/sampling_logp_difference/max": 0.8032641410827637, "sampling/sampling_logp_difference/mean": 0.008658221922814846, "step": 3095 }, { "clip_ratio/high_max": 0.006390700465999544, "clip_ratio/high_mean": 0.006390700465999544, "clip_ratio/low_mean": 0.0025510203558951616, "clip_ratio/low_min": 0.0025510203558951616, "clip_ratio/region_mean": 0.008941720821894705, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.875, "completions/mean_terminated_length": 97.875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.03362013725563884, "epoch": 0.12435233160621761, "frac_reward_zero_std": 0.0, "grad_norm": 0.717937171459198, "learning_rate": 6.212121212121212e-07, "loss": -0.0004, "num_tokens": 7039952.0, "reward": 0.9993786811828613, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993786811828613, "reward_meter_std": 8.659059676574543e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.658833394292742e-05, "reward_total_composite_mean": 0.9993786811828613, "reward_total_composite_std": 8.659059676574543e-05, "reward_total_mean": 0.9993786811828613, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993786811828613, "rewards/meter/std": 8.659059676574543e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993786811828613, "rewards/total_composite/std": 8.659059676574543e-05, "sampling/importance_sampling_ratio/max": 1.574755072593689, "sampling/importance_sampling_ratio/mean": 0.9991164803504944, "sampling/importance_sampling_ratio/min": 0.3799818456172943, "sampling/sampling_logp_difference/max": 0.9676318168640137, "sampling/sampling_logp_difference/mean": 0.006720404606312513, "step": 3096 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.005681818351149559, "clip_ratio/low_min": 0.005681818351149559, "clip_ratio/region_mean": 0.007547489949502051, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.375, "completions/mean_terminated_length": 66.375, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.05762590793892741, "epoch": 0.12439249708800257, "frac_reward_zero_std": 0.0, "grad_norm": 0.17411062121391296, "learning_rate": 6.181818181818182e-07, "loss": -0.0005, "num_tokens": 7041771.0, "reward": 0.9981486797332764, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981486797332764, "reward_meter_std": 1.2580998372868635e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.2579364010889549e-05, "reward_total_composite_mean": 0.9981486797332764, "reward_total_composite_std": 1.2580998372868635e-05, "reward_total_mean": 0.9981486797332764, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981486797332764, "rewards/meter/std": 1.2580998372868635e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981486797332764, "rewards/total_composite/std": 1.2580998372868635e-05, "sampling/importance_sampling_ratio/max": 1.580132246017456, "sampling/importance_sampling_ratio/mean": 1.0030303001403809, "sampling/importance_sampling_ratio/min": 0.5710699558258057, "sampling/sampling_logp_difference/max": 0.5602436065673828, "sampling/sampling_logp_difference/mean": 0.00830040592700243, "step": 3097 }, { "clip_ratio/high_max": 0.025927380891516805, "clip_ratio/high_mean": 0.025927380891516805, "clip_ratio/low_mean": 0.006949802860617638, "clip_ratio/low_min": 0.006949802860617638, "clip_ratio/region_mean": 0.03287718375213444, "completions/clipped_ratio": 0.0, "completions/max_length": 310.0, "completions/max_terminated_length": 310.0, "completions/mean_length": 283.125, "completions/mean_terminated_length": 283.125, "completions/min_length": 262.0, "completions/min_terminated_length": 262.0, "entropy": 0.522564172744751, "epoch": 0.12443266256978752, "frac_reward_zero_std": 0.0, "grad_norm": 2.0594077110290527, "learning_rate": 6.151515151515152e-07, "loss": -0.0254, "num_tokens": 7045692.0, "reward": 0.7604771852493286, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_meter_mean": 0.9590431451797485, "reward_meter_std": 0.07114315778017044, "reward_repeat_penalty_mean": 0.9272365570068359, "reward_repeat_penalty_std": 0.04068033769726753, "reward_std": 0.31678304076194763, "reward_total_composite_mean": 0.7604771852493286, "reward_total_composite_std": 0.31678301095962524, "reward_total_mean": 0.7604771852493286, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/meter/mean": 0.9590431451797485, "rewards/meter/std": 0.07114315778017044, "rewards/repeat_penalty/mean": 0.9272365570068359, "rewards/repeat_penalty/std": 0.04068033769726753, "rewards/total_composite/mean": 0.7604771852493286, "rewards/total_composite/std": 0.31678301095962524, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0093165636062622, "sampling/importance_sampling_ratio/min": 0.22849158942699432, "sampling/sampling_logp_difference/max": 1.4762558937072754, "sampling/sampling_logp_difference/mean": 0.05060786381363869, "step": 3098 }, { "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/low_mean": 0.005138941807672381, "clip_ratio/low_min": 0.005138941807672381, "clip_ratio/region_mean": 0.008660068502649665, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 71.375, "completions/mean_terminated_length": 71.375, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.07044319622218609, "epoch": 0.12447282805157248, "frac_reward_zero_std": 0.0, "grad_norm": 0.2977856993675232, "learning_rate": 6.121212121212121e-07, "loss": 0.0008, "num_tokens": 7047543.0, "reward": 0.9994341135025024, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994341135025024, "reward_meter_std": 2.3516733563155867e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.3526572476839647e-05, "reward_total_composite_mean": 0.9994341135025024, "reward_total_composite_std": 2.3516733563155867e-05, "reward_total_mean": 0.9994341135025024, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994341135025024, "rewards/meter/std": 2.3516733563155867e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994341135025024, "rewards/total_composite/std": 2.3516733563155867e-05, "sampling/importance_sampling_ratio/max": 1.2391035556793213, "sampling/importance_sampling_ratio/mean": 0.999859631061554, "sampling/importance_sampling_ratio/min": 0.3569888472557068, "sampling/sampling_logp_difference/max": 1.0300507545471191, "sampling/sampling_logp_difference/mean": 0.010309785604476929, "step": 3099 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0007583037222502753, "epoch": 0.12451299353335743, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.090909090909092e-07, "loss": 0.0, "num_tokens": 7049375.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0023247003555298, "sampling/importance_sampling_ratio/mean": 1.0000481605529785, "sampling/importance_sampling_ratio/min": 0.9947122931480408, "sampling/sampling_logp_difference/max": 0.0053017400205135345, "sampling/sampling_logp_difference/mean": 8.230307139456272e-05, "step": 3100 }, { "epoch": 0.12451299353335743, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.019230769230769232, "eval_completions/max_length": 415.0, "eval_completions/max_terminated_length": 392.53846153846155, "eval_completions/mean_length": 210.7403846153846, "eval_completions/mean_terminated_length": 204.17994689941406, "eval_completions/min_length": 61.07692307692308, "eval_completions/min_terminated_length": 61.07692307692308, "eval_entropy": 0.37355728218188655, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 7049375.0, "eval_reward": 0.7422462472548852, "eval_reward_arabic_clean_mean": 0.9903846153846154, "eval_reward_arabic_clean_std": 0.027196414195574246, "eval_reward_count_adherence_mean": 0.9585646253365737, "eval_reward_count_adherence_std": 0.065789727637401, "eval_reward_meter_mean": 0.8045936226844788, "eval_reward_meter_std": 0.3283666269137309, "eval_reward_repeat_penalty_mean": 0.9568017308528607, "eval_reward_repeat_penalty_std": 0.065969582217244, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7422462472548852, "eval_reward_total_composite_std": 0.3302810134795996, "eval_reward_total_mean": 0.7422462472548852, "eval_rewards/arabic_clean/mean": 0.9903846153846154, "eval_rewards/arabic_clean/std": 0.027196414195574246, "eval_rewards/count_adherence/mean": 0.9585646253365737, "eval_rewards/count_adherence/std": 0.065789727637401, "eval_rewards/meter/mean": 0.8045936226844788, "eval_rewards/meter/std": 0.3283666269137309, "eval_rewards/repeat_penalty/mean": 0.9568017308528607, "eval_rewards/repeat_penalty/std": 0.065969582217244, "eval_rewards/total_composite/mean": 0.7422462472548852, "eval_rewards/total_composite/std": 0.3302810134795996, "eval_runtime": 77.6639, "eval_samples_per_second": 1.339, "eval_sampling/importance_sampling_ratio/max": 1.5006029880963838, "eval_sampling/importance_sampling_ratio/mean": 1.009482246178847, "eval_sampling/importance_sampling_ratio/min": 0.3349529756949498, "eval_sampling/sampling_logp_difference/max": 1.120217965199397, "eval_sampling/sampling_logp_difference/mean": 0.03276929044379638, "eval_steps_per_second": 0.167, "step": 3100 }, { "clip_ratio/high_max": 0.03715289803221822, "clip_ratio/high_mean": 0.03715289803221822, "clip_ratio/low_mean": 0.005714374827221036, "clip_ratio/low_min": 0.005714374827221036, "clip_ratio/region_mean": 0.042867272859439254, "completions/clipped_ratio": 0.0, "completions/max_length": 223.0, "completions/max_terminated_length": 223.0, "completions/mean_length": 209.375, "completions/mean_terminated_length": 209.375, "completions/min_length": 198.0, "completions/min_terminated_length": 198.0, "entropy": 0.42885980382561684, "epoch": 0.12455315901514238, "frac_reward_zero_std": 0.0, "grad_norm": 2.886378765106201, "learning_rate": 6.060606060606061e-07, "loss": 0.0255, "num_tokens": 7052434.0, "reward": 0.9347763061523438, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9796897768974304, "reward_meter_std": 0.013765236362814903, "reward_repeat_penalty_mean": 0.9545454978942871, "reward_repeat_penalty_std": 0.0971859022974968, "reward_std": 0.0924786925315857, "reward_total_composite_mean": 0.9347763061523438, "reward_total_composite_std": 0.0924786925315857, "reward_total_mean": 0.9347763061523438, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9796897768974304, "rewards/meter/std": 0.013765236362814903, "rewards/repeat_penalty/mean": 0.9545454978942871, "rewards/repeat_penalty/std": 0.0971859022974968, "rewards/total_composite/mean": 0.9347763061523438, "rewards/total_composite/std": 0.0924786925315857, "sampling/importance_sampling_ratio/max": 1.8678045272827148, "sampling/importance_sampling_ratio/mean": 1.0072675943374634, "sampling/importance_sampling_ratio/min": 0.40461668372154236, "sampling/sampling_logp_difference/max": 0.9048151969909668, "sampling/sampling_logp_difference/mean": 0.03803591802716255, "step": 3101 }, { "clip_ratio/high_max": 0.005800189450383186, "clip_ratio/high_mean": 0.005800189450383186, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005800189450383186, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.5, "completions/mean_terminated_length": 65.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.14634791854768991, "epoch": 0.12459332449692734, "frac_reward_zero_std": 0.0, "grad_norm": 3.9274234771728516, "learning_rate": 6.03030303030303e-07, "loss": 0.0064, "num_tokens": 7054118.0, "reward": 0.9924110770225525, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9924110770225525, "reward_meter_std": 0.00258190231397748, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0025818964932113886, "reward_total_composite_mean": 0.9924110770225525, "reward_total_composite_std": 0.00258190231397748, "reward_total_mean": 0.9924110770225525, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9924110770225525, "rewards/meter/std": 0.00258190231397748, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9924110770225525, "rewards/total_composite/std": 0.00258190231397748, "sampling/importance_sampling_ratio/max": 1.3347103595733643, "sampling/importance_sampling_ratio/mean": 1.002340316772461, "sampling/importance_sampling_ratio/min": 0.30476200580596924, "sampling/sampling_logp_difference/max": 1.1882240772247314, "sampling/sampling_logp_difference/mean": 0.023297905921936035, "step": 3102 }, { "clip_ratio/high_max": 0.03492570295929909, "clip_ratio/high_mean": 0.03492570295929909, "clip_ratio/low_mean": 0.017914682626724243, "clip_ratio/low_min": 0.017914682626724243, "clip_ratio/region_mean": 0.05284038558602333, "completions/clipped_ratio": 0.0, "completions/max_length": 88.0, "completions/max_terminated_length": 88.0, "completions/mean_length": 85.875, "completions/mean_terminated_length": 85.875, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.26804009079933167, "epoch": 0.12463348997871229, "frac_reward_zero_std": 0.0, "grad_norm": 5.046589374542236, "learning_rate": 6.000000000000001e-07, "loss": 0.0107, "num_tokens": 7056293.0, "reward": 0.930683970451355, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9477604627609253, "reward_meter_std": 0.021915482357144356, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.04980317875742912, "reward_total_composite_mean": 0.930683970451355, "reward_total_composite_std": 0.04980318620800972, "reward_total_mean": 0.930683970451355, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9477604627609253, "rewards/meter/std": 0.021915482357144356, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.930683970451355, "rewards/total_composite/std": 0.04980318620800972, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.993628203868866, "sampling/importance_sampling_ratio/min": 0.15559034049510956, "sampling/sampling_logp_difference/max": 1.8605287075042725, "sampling/sampling_logp_difference/mean": 0.06328704953193665, "step": 3103 }, { "clip_ratio/high_max": 0.00967782223597169, "clip_ratio/high_mean": 0.00967782223597169, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.00967782223597169, "completions/clipped_ratio": 0.0, "completions/max_length": 78.0, "completions/max_terminated_length": 78.0, "completions/mean_length": 77.625, "completions/mean_terminated_length": 77.625, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.18993182107806206, "epoch": 0.12467365546049725, "frac_reward_zero_std": 0.0, "grad_norm": 3.8720288276672363, "learning_rate": 5.96969696969697e-07, "loss": 0.002, "num_tokens": 7058098.0, "reward": 0.9990447163581848, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990447163581848, "reward_meter_std": 0.0004974787007085979, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004974827752448618, "reward_total_composite_mean": 0.9990447163581848, "reward_total_composite_std": 0.0004974787007085979, "reward_total_mean": 0.9990447163581848, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990447163581848, "rewards/meter/std": 0.0004974787007085979, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990447163581848, "rewards/total_composite/std": 0.0004974787007085979, "sampling/importance_sampling_ratio/max": 1.4500758647918701, "sampling/importance_sampling_ratio/mean": 1.0045377016067505, "sampling/importance_sampling_ratio/min": 0.4161592125892639, "sampling/sampling_logp_difference/max": 0.8766874074935913, "sampling/sampling_logp_difference/mean": 0.019970104098320007, "step": 3104 }, { "clip_ratio/high_max": 0.02611940260976553, "clip_ratio/high_mean": 0.02611940260976553, "clip_ratio/low_mean": 0.007753314450383186, "clip_ratio/low_min": 0.007753314450383186, "clip_ratio/region_mean": 0.033872717060148716, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.5, "completions/mean_terminated_length": 66.5, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.13373739924281836, "epoch": 0.1247138209422822, "frac_reward_zero_std": 0.0, "grad_norm": 6.449222087860107, "learning_rate": 5.93939393939394e-07, "loss": -0.0052, "num_tokens": 7059998.0, "reward": 0.9450548887252808, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9450548887252808, "reward_meter_std": 0.015829402953386307, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.015829408541321754, "reward_total_composite_mean": 0.9450548887252808, "reward_total_composite_std": 0.015829402953386307, "reward_total_mean": 0.9450548887252808, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9450548887252808, "rewards/meter/std": 0.015829402953386307, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9450548887252808, "rewards/total_composite/std": 0.015829402953386307, "sampling/importance_sampling_ratio/max": 1.7957602739334106, "sampling/importance_sampling_ratio/mean": 1.0009596347808838, "sampling/importance_sampling_ratio/min": 0.19941234588623047, "sampling/sampling_logp_difference/max": 1.6123805046081543, "sampling/sampling_logp_difference/mean": 0.03087073564529419, "step": 3105 }, { "clip_ratio/high_max": 0.0121077821822837, "clip_ratio/high_mean": 0.0121077821822837, "clip_ratio/low_mean": 0.004859996202867478, "clip_ratio/low_min": 0.004859996202867478, "clip_ratio/region_mean": 0.016967778385151178, "completions/clipped_ratio": 0.0, "completions/max_length": 157.0, "completions/max_terminated_length": 157.0, "completions/mean_length": 155.0, "completions/mean_terminated_length": 155.0, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "entropy": 0.3304919973015785, "epoch": 0.12475398642406715, "frac_reward_zero_std": 0.0, "grad_norm": 1.8398211002349854, "learning_rate": 5.90909090909091e-07, "loss": 0.0012, "num_tokens": 7062542.0, "reward": 0.9989175200462341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989175200462341, "reward_meter_std": 0.00033335990156047046, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0003333572531118989, "reward_total_composite_mean": 0.9989175200462341, "reward_total_composite_std": 0.00033335990156047046, "reward_total_mean": 0.9989175200462341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989175200462341, "rewards/meter/std": 0.00033335990156047046, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989175200462341, "rewards/total_composite/std": 0.00033335990156047046, "sampling/importance_sampling_ratio/max": 1.7075154781341553, "sampling/importance_sampling_ratio/mean": 1.0071746110916138, "sampling/importance_sampling_ratio/min": 0.0892224907875061, "sampling/sampling_logp_difference/max": 2.4166221618652344, "sampling/sampling_logp_difference/mean": 0.03886400908231735, "step": 3106 }, { "clip_ratio/high_max": 0.014827882871031761, "clip_ratio/high_mean": 0.014827882871031761, "clip_ratio/low_mean": 0.01397907012142241, "clip_ratio/low_min": 0.01397907012142241, "clip_ratio/region_mean": 0.02880695299245417, "completions/clipped_ratio": 0.0, "completions/max_length": 426.0, "completions/max_terminated_length": 426.0, "completions/mean_length": 403.875, "completions/mean_terminated_length": 403.875, "completions/min_length": 383.0, "completions/min_terminated_length": 383.0, "entropy": 0.4669685959815979, "epoch": 0.12479415190585211, "frac_reward_zero_std": 0.0, "grad_norm": 1.4729881286621094, "learning_rate": 5.878787878787879e-07, "loss": -0.0223, "num_tokens": 7067381.0, "reward": 0.7777890563011169, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7980769276618958, "reward_count_adherence_std": 0.039811473339796066, "reward_meter_mean": 0.9989637136459351, "reward_meter_std": 0.00037519802572205663, "reward_repeat_penalty_mean": 0.9746710062026978, "reward_repeat_penalty_std": 0.038055676966905594, "reward_std": 0.061021264642477036, "reward_total_composite_mean": 0.7777890563011169, "reward_total_composite_std": 0.061021268367767334, "reward_total_mean": 0.7777890563011169, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7980769276618958, "rewards/count_adherence/std": 0.039811473339796066, "rewards/meter/mean": 0.9989637136459351, "rewards/meter/std": 0.00037519802572205663, "rewards/repeat_penalty/mean": 0.9746710062026978, "rewards/repeat_penalty/std": 0.038055676966905594, "rewards/total_composite/mean": 0.7777890563011169, "rewards/total_composite/std": 0.061021268367767334, "sampling/importance_sampling_ratio/max": 1.9504014253616333, "sampling/importance_sampling_ratio/mean": 1.0116568803787231, "sampling/importance_sampling_ratio/min": 0.20386646687984467, "sampling/sampling_logp_difference/max": 1.5902900695800781, "sampling/sampling_logp_difference/mean": 0.04732804745435715, "step": 3107 }, { "clip_ratio/high_max": 0.0067204301012679935, "clip_ratio/high_mean": 0.0067204301012679935, "clip_ratio/low_mean": 0.003989361692219973, "clip_ratio/low_min": 0.003989361692219973, "clip_ratio/region_mean": 0.010709791793487966, "completions/clipped_ratio": 0.0, "completions/max_length": 94.0, "completions/max_terminated_length": 94.0, "completions/mean_length": 93.125, "completions/mean_terminated_length": 93.125, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.0752844586968422, "epoch": 0.12483431738763706, "frac_reward_zero_std": 0.0, "grad_norm": 0.7012980580329895, "learning_rate": 5.848484848484849e-07, "loss": 0.0018, "num_tokens": 7069454.0, "reward": 0.9972705841064453, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972705841064453, "reward_meter_std": 0.0013248658506199718, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0013248566538095474, "reward_total_composite_mean": 0.9972705841064453, "reward_total_composite_std": 0.0013248658506199718, "reward_total_mean": 0.9972705841064453, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972705841064453, "rewards/meter/std": 0.0013248658506199718, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972705841064453, "rewards/total_composite/std": 0.0013248658506199718, "sampling/importance_sampling_ratio/max": 1.6383498907089233, "sampling/importance_sampling_ratio/mean": 1.004794716835022, "sampling/importance_sampling_ratio/min": 0.6166864037513733, "sampling/sampling_logp_difference/max": 0.49368953704833984, "sampling/sampling_logp_difference/mean": 0.009307874366641045, "step": 3108 }, { "clip_ratio/high_max": 0.029623867012560368, "clip_ratio/high_mean": 0.029623867012560368, "clip_ratio/low_mean": 0.011014241958037019, "clip_ratio/low_min": 0.011014241958037019, "clip_ratio/region_mean": 0.040638108970597386, "completions/clipped_ratio": 0.0, "completions/max_length": 178.0, "completions/max_terminated_length": 178.0, "completions/mean_length": 172.125, "completions/mean_terminated_length": 172.125, "completions/min_length": 165.0, "completions/min_terminated_length": 165.0, "entropy": 0.37529650889337063, "epoch": 0.12487448286942202, "frac_reward_zero_std": 0.0, "grad_norm": 3.0335049629211426, "learning_rate": 5.818181818181819e-07, "loss": 0.0016, "num_tokens": 7072207.0, "reward": 0.9342338442802429, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.961860179901123, "reward_meter_std": 0.08823229372501373, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.09130982309579849, "reward_total_composite_mean": 0.9342338442802429, "reward_total_composite_std": 0.0913098156452179, "reward_total_mean": 0.9342338442802429, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.961860179901123, "rewards/meter/std": 0.08823229372501373, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9342338442802429, "rewards/total_composite/std": 0.0913098156452179, "sampling/importance_sampling_ratio/max": 1.8755393028259277, "sampling/importance_sampling_ratio/mean": 1.0083178281784058, "sampling/importance_sampling_ratio/min": 0.12143982201814651, "sampling/sampling_logp_difference/max": 2.1083364486694336, "sampling/sampling_logp_difference/mean": 0.03732313588261604, "step": 3109 }, { "clip_ratio/high_max": 0.03839668887667358, "clip_ratio/high_mean": 0.03839668887667358, "clip_ratio/low_mean": 0.004545454401522875, "clip_ratio/low_min": 0.004545454401522875, "clip_ratio/region_mean": 0.042942143278196454, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 166.0, "completions/mean_terminated_length": 166.0, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.3597927987575531, "epoch": 0.12491464835120697, "frac_reward_zero_std": 0.0, "grad_norm": 2.6455962657928467, "learning_rate": 5.787878787878789e-07, "loss": 0.0026, "num_tokens": 7074959.0, "reward": 0.9984124898910522, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984124898910522, "reward_meter_std": 0.0009795344667509198, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009795371443033218, "reward_total_composite_mean": 0.9984124898910522, "reward_total_composite_std": 0.0009795344667509198, "reward_total_mean": 0.9984124898910522, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984124898910522, "rewards/meter/std": 0.0009795344667509198, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984124898910522, "rewards/total_composite/std": 0.0009795344667509198, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0051828622817993, "sampling/importance_sampling_ratio/min": 0.02957269363105297, "sampling/sampling_logp_difference/max": 3.5209038257598877, "sampling/sampling_logp_difference/mean": 0.04640757292509079, "step": 3110 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0017129705229308456, "epoch": 0.12495481383299192, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.757575757575758e-07, "loss": 0.0, "num_tokens": 7076575.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0025551319122314, "sampling/importance_sampling_ratio/mean": 1.0002080202102661, "sampling/importance_sampling_ratio/min": 0.9997379779815674, "sampling/sampling_logp_difference/max": 0.0025518755428493023, "sampling/sampling_logp_difference/mean": 0.00021090851805638522, "step": 3111 }, { "clip_ratio/high_max": 0.026788415852934122, "clip_ratio/high_mean": 0.026788415852934122, "clip_ratio/low_mean": 0.022525052540004253, "clip_ratio/low_min": 0.022525052540004253, "clip_ratio/region_mean": 0.049313468392938375, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 303.0, "completions/mean_terminated_length": 303.0, "completions/min_length": 286.0, "completions/min_terminated_length": 286.0, "entropy": 0.5345265790820122, "epoch": 0.12499497931477688, "frac_reward_zero_std": 0.0, "grad_norm": 2.5759596824645996, "learning_rate": 5.727272727272728e-07, "loss": 0.0147, "num_tokens": 7080735.0, "reward": 0.9937278628349304, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9937278628349304, "reward_meter_std": 0.004818979650735855, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004818981513381004, "reward_total_composite_mean": 0.9937278628349304, "reward_total_composite_std": 0.004818979650735855, "reward_total_mean": 0.9937278628349304, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9937278628349304, "rewards/meter/std": 0.004818979650735855, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9937278628349304, "rewards/total_composite/std": 0.004818979650735855, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0059781074523926, "sampling/importance_sampling_ratio/min": 0.0015555039281025529, "sampling/sampling_logp_difference/max": 6.46595573425293, "sampling/sampling_logp_difference/mean": 0.0645064041018486, "step": 3112 }, { "clip_ratio/high_max": 0.006097560748457909, "clip_ratio/high_mean": 0.006097560748457909, "clip_ratio/low_mean": 0.0062500000931322575, "clip_ratio/low_min": 0.0062500000931322575, "clip_ratio/region_mean": 0.012347560841590166, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 40.625, "completions/mean_terminated_length": 40.625, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.09460031241178513, "epoch": 0.12503514479656183, "frac_reward_zero_std": 0.0, "grad_norm": 0.42850980162620544, "learning_rate": 5.696969696969698e-07, "loss": -0.0003, "num_tokens": 7082156.0, "reward": 0.998562216758728, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998562216758728, "reward_meter_std": 0.0005881982506252825, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0005882052937522531, "reward_total_composite_mean": 0.998562216758728, "reward_total_composite_std": 0.0005881982506252825, "reward_total_mean": 0.998562216758728, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998562216758728, "rewards/meter/std": 0.0005881982506252825, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998562216758728, "rewards/total_composite/std": 0.0005881982506252825, "sampling/importance_sampling_ratio/max": 1.4733530282974243, "sampling/importance_sampling_ratio/mean": 1.0016939640045166, "sampling/importance_sampling_ratio/min": 0.47915592789649963, "sampling/sampling_logp_difference/max": 0.7357292175292969, "sampling/sampling_logp_difference/mean": 0.009273692034184933, "step": 3113 }, { "clip_ratio/high_max": 0.008981169550679624, "clip_ratio/high_mean": 0.008981169550679624, "clip_ratio/low_mean": 0.0013157895300537348, "clip_ratio/low_min": 0.0013157895300537348, "clip_ratio/region_mean": 0.010296959080733359, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.25, "completions/mean_terminated_length": 97.25, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.10958225093781948, "epoch": 0.12507531027834679, "frac_reward_zero_std": 0.0, "grad_norm": 1.817691683769226, "learning_rate": 5.666666666666667e-07, "loss": -0.0061, "num_tokens": 7084326.0, "reward": 0.9978147745132446, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978147745132446, "reward_meter_std": 0.00044735919800587, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004473614681046456, "reward_total_composite_mean": 0.9978147745132446, "reward_total_composite_std": 0.00044735919800587, "reward_total_mean": 0.9978147745132446, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978147745132446, "rewards/meter/std": 0.00044735919800587, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9978147745132446, "rewards/total_composite/std": 0.00044735919800587, "sampling/importance_sampling_ratio/max": 1.3954778909683228, "sampling/importance_sampling_ratio/mean": 1.0041371583938599, "sampling/importance_sampling_ratio/min": 0.3762938380241394, "sampling/sampling_logp_difference/max": 0.9773850440979004, "sampling/sampling_logp_difference/mean": 0.013976364396512508, "step": 3114 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.004098360426723957, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.02565961377695203, "epoch": 0.12511547576013174, "frac_reward_zero_std": 0.0, "grad_norm": 0.015785139054059982, "learning_rate": 5.636363636363638e-07, "loss": 0.0004, "num_tokens": 7086062.0, "reward": 0.9973392486572266, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973392486572266, "reward_meter_std": 4.345127990745823e-07, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.280403231859964e-07, "reward_total_composite_mean": 0.9973392486572266, "reward_total_composite_std": 4.345127990745823e-07, "reward_total_mean": 0.9973392486572266, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973392486572266, "rewards/meter/std": 4.345127990745823e-07, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973392486572266, "rewards/total_composite/std": 4.345127990745823e-07, "sampling/importance_sampling_ratio/max": 1.2580742835998535, "sampling/importance_sampling_ratio/mean": 0.9995636940002441, "sampling/importance_sampling_ratio/min": 0.5760312080383301, "sampling/sampling_logp_difference/max": 0.5515934228897095, "sampling/sampling_logp_difference/mean": 0.004269226919859648, "step": 3115 }, { "clip_ratio/high_max": 0.013212713180109859, "clip_ratio/high_mean": 0.013212713180109859, "clip_ratio/low_mean": 0.0044667941983789206, "clip_ratio/low_min": 0.0044667941983789206, "clip_ratio/region_mean": 0.01767950737848878, "completions/clipped_ratio": 0.125, "completions/max_length": 512.0, "completions/max_terminated_length": 509.0, "completions/mean_length": 490.75, "completions/mean_terminated_length": 487.71429443359375, "completions/min_length": 461.0, "completions/min_terminated_length": 461.0, "entropy": 0.3404971808195114, "epoch": 0.1251556412419167, "frac_reward_zero_std": 0.0, "grad_norm": 1.0452237129211426, "learning_rate": 5.606060606060607e-07, "loss": 0.0232, "num_tokens": 7091420.0, "reward": 0.8266471028327942, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8515625, "reward_count_adherence_std": 0.032346826046705246, "reward_meter_mean": 0.9985963106155396, "reward_meter_std": 0.0007045524544082582, "reward_repeat_penalty_mean": 0.9714829921722412, "reward_repeat_penalty_std": 0.027092305943369865, "reward_std": 0.049881432205438614, "reward_total_composite_mean": 0.8266471028327942, "reward_total_composite_std": 0.049881432205438614, "reward_total_mean": 0.8266471028327942, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8515625, "rewards/count_adherence/std": 0.032346826046705246, "rewards/meter/mean": 0.9985963106155396, "rewards/meter/std": 0.0007045524544082582, "rewards/repeat_penalty/mean": 0.9714829921722412, "rewards/repeat_penalty/std": 0.027092305943369865, "rewards/total_composite/mean": 0.8266471028327942, "rewards/total_composite/std": 0.049881432205438614, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0090035200119019, "sampling/importance_sampling_ratio/min": 0.2545214593410492, "sampling/sampling_logp_difference/max": 1.3683700561523438, "sampling/sampling_logp_difference/mean": 0.034601159393787384, "step": 3116 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 34.0, "completions/mean_terminated_length": 34.0, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.03289771918207407, "epoch": 0.12519580672370165, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.575757575757576e-07, "loss": 0.0, "num_tokens": 7092924.0, "reward": 0.9923644065856934, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9923644065856934, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9923644065856934, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9923644065856934, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9923644065856934, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9923644065856934, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.1301469802856445, "sampling/importance_sampling_ratio/mean": 1.0040427446365356, "sampling/importance_sampling_ratio/min": 0.9824780821800232, "sampling/sampling_logp_difference/max": 0.12234780192375183, "sampling/sampling_logp_difference/mean": 0.004286302253603935, "step": 3117 }, { "clip_ratio/high_max": 0.008788066916167736, "clip_ratio/high_mean": 0.008788066916167736, "clip_ratio/low_mean": 0.008492077584378421, "clip_ratio/low_min": 0.008492077584378421, "clip_ratio/region_mean": 0.017280144500546157, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 354.625, "completions/mean_terminated_length": 354.625, "completions/min_length": 350.0, "completions/min_terminated_length": 350.0, "entropy": 0.3398951441049576, "epoch": 0.1252359722054866, "frac_reward_zero_std": 0.0, "grad_norm": 1.1702804565429688, "learning_rate": 5.545454545454547e-07, "loss": 0.0017, "num_tokens": 7097281.0, "reward": 0.8780869841575623, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9090909361839294, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987514019012451, "reward_meter_std": 0.00020078917441423982, "reward_repeat_penalty_mean": 0.9671052694320679, "reward_repeat_penalty_std": 0.027239417657256126, "reward_std": 0.02465316839516163, "reward_total_composite_mean": 0.8780869841575623, "reward_total_composite_std": 0.024653173983097076, "reward_total_mean": 0.8780869841575623, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9090909361839294, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987514019012451, "rewards/meter/std": 0.00020078917441423982, "rewards/repeat_penalty/mean": 0.9671052694320679, "rewards/repeat_penalty/std": 0.027239417657256126, "rewards/total_composite/mean": 0.8780869841575623, "rewards/total_composite/std": 0.024653173983097076, "sampling/importance_sampling_ratio/max": 1.7627272605895996, "sampling/importance_sampling_ratio/mean": 1.0095971822738647, "sampling/importance_sampling_ratio/min": 0.2586033344268799, "sampling/sampling_logp_difference/max": 1.3524599075317383, "sampling/sampling_logp_difference/mean": 0.029794396832585335, "step": 3118 }, { "clip_ratio/high_max": 0.004098360426723957, "clip_ratio/high_mean": 0.004098360426723957, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.004098360426723957, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 60.875, "completions/mean_terminated_length": 60.875, "completions/min_length": 60.0, "completions/min_terminated_length": 60.0, "entropy": 0.02145107788965106, "epoch": 0.12527613768727155, "frac_reward_zero_std": 0.0, "grad_norm": 5.017796516418457, "learning_rate": 5.515151515151516e-07, "loss": -0.0005, "num_tokens": 7098968.0, "reward": 0.9972956776618958, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9972956776618958, "reward_meter_std": 0.0001135577549575828, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001135657585109584, "reward_total_composite_mean": 0.9972956776618958, "reward_total_composite_std": 0.0001135577549575828, "reward_total_mean": 0.9972956776618958, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9972956776618958, "rewards/meter/std": 0.0001135577549575828, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9972956776618958, "rewards/total_composite/std": 0.0001135577549575828, "sampling/importance_sampling_ratio/max": 1.1826646327972412, "sampling/importance_sampling_ratio/mean": 0.9997225999832153, "sampling/importance_sampling_ratio/min": 0.2997862994670868, "sampling/sampling_logp_difference/max": 1.2046854496002197, "sampling/sampling_logp_difference/mean": 0.005037278868257999, "step": 3119 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0014699531602673233, "epoch": 0.1253163031690565, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.484848484848485e-07, "loss": 0.0, "num_tokens": 7100936.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0016909837722778, "sampling/importance_sampling_ratio/mean": 1.0001810789108276, "sampling/importance_sampling_ratio/min": 0.998882532119751, "sampling/sampling_logp_difference/max": 0.0016895392909646034, "sampling/sampling_logp_difference/mean": 0.00018837934476323426, "step": 3120 }, { "clip_ratio/high_max": 0.02431198745034635, "clip_ratio/high_mean": 0.02431198745034635, "clip_ratio/low_mean": 0.007510315626859665, "clip_ratio/low_min": 0.007510315626859665, "clip_ratio/region_mean": 0.031822303077206016, "completions/clipped_ratio": 0.0, "completions/max_length": 394.0, "completions/max_terminated_length": 394.0, "completions/mean_length": 354.625, "completions/mean_terminated_length": 354.625, "completions/min_length": 342.0, "completions/min_terminated_length": 342.0, "entropy": 0.47248122096061707, "epoch": 0.12535646865084146, "frac_reward_zero_std": 0.0, "grad_norm": 1.84979248046875, "learning_rate": 5.454545454545455e-07, "loss": 0.0239, "num_tokens": 7105325.0, "reward": 0.6744744777679443, "reward_arabic_clean_mean": 0.75, "reward_arabic_clean_std": 0.4629100561141968, "reward_count_adherence_mean": 0.9124999642372131, "reward_count_adherence_std": 0.0353553481400013, "reward_meter_mean": 0.9991722106933594, "reward_meter_std": 0.0001384944043820724, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.4162946939468384, "reward_total_composite_mean": 0.6744744777679443, "reward_total_composite_std": 0.41629472374916077, "reward_total_mean": 0.6744744777679443, "rewards/arabic_clean/mean": 0.75, "rewards/arabic_clean/std": 0.4629100561141968, "rewards/count_adherence/mean": 0.9124999642372131, "rewards/count_adherence/std": 0.0353553481400013, "rewards/meter/mean": 0.9991722106933594, "rewards/meter/std": 0.0001384944043820724, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.6744744777679443, "rewards/total_composite/std": 0.41629472374916077, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0077815055847168, "sampling/importance_sampling_ratio/min": 0.0938795730471611, "sampling/sampling_logp_difference/max": 2.3657424449920654, "sampling/sampling_logp_difference/mean": 0.04996606335043907, "step": 3121 }, { "clip_ratio/high_max": 0.006147540640085936, "clip_ratio/high_mean": 0.006147540640085936, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006147540640085936, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.019400187535211444, "epoch": 0.12539663413262642, "frac_reward_zero_std": 0.0, "grad_norm": 0.5849205851554871, "learning_rate": 5.424242424242425e-07, "loss": 0.0001, "num_tokens": 7107085.0, "reward": 0.9973349571228027, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973349571228027, "reward_meter_std": 1.3741479051532224e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.3751973710895982e-05, "reward_total_composite_mean": 0.9973349571228027, "reward_total_composite_std": 1.3741479051532224e-05, "reward_total_mean": 0.9973349571228027, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973349571228027, "rewards/meter/std": 1.3741479051532224e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973349571228027, "rewards/total_composite/std": 1.3741479051532224e-05, "sampling/importance_sampling_ratio/max": 1.0885894298553467, "sampling/importance_sampling_ratio/mean": 0.9968710541725159, "sampling/importance_sampling_ratio/min": 0.44264963269233704, "sampling/sampling_logp_difference/max": 0.814976692199707, "sampling/sampling_logp_difference/mean": 0.005648355465382338, "step": 3122 }, { "clip_ratio/high_max": 0.02493911876808852, "clip_ratio/high_mean": 0.02493911876808852, "clip_ratio/low_mean": 0.007426470518112183, "clip_ratio/low_min": 0.007426470518112183, "clip_ratio/region_mean": 0.0323655892862007, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 100.5, "completions/mean_terminated_length": 100.5, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.1784425787627697, "epoch": 0.12543679961441137, "frac_reward_zero_std": 0.0, "grad_norm": 2.6911187171936035, "learning_rate": 5.393939393939395e-07, "loss": 0.0044, "num_tokens": 7109065.0, "reward": 0.9990350008010864, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990350008010864, "reward_meter_std": 0.0004222920979373157, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004223074938636273, "reward_total_composite_mean": 0.9990350008010864, "reward_total_composite_std": 0.0004222920979373157, "reward_total_mean": 0.9990350008010864, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990350008010864, "rewards/meter/std": 0.0004222920979373157, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990350008010864, "rewards/total_composite/std": 0.0004222920979373157, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0030157566070557, "sampling/importance_sampling_ratio/min": 0.5429390072822571, "sampling/sampling_logp_difference/max": 0.8458013534545898, "sampling/sampling_logp_difference/mean": 0.025933468714356422, "step": 3123 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.12547696509619632, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 5.363636363636364e-07, "loss": 0.0, "num_tokens": 7110721.0, "reward": 0.6985058784484863, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7361111044883728, "reward_count_adherence_std": 0.0257172379642725, "reward_meter_mean": 0.9872919917106628, "reward_meter_std": 0.03354829549789429, "reward_repeat_penalty_mean": 0.9615384340286255, "reward_repeat_penalty_std": 0.029074188321828842, "reward_std": 0.03456050530076027, "reward_total_composite_mean": 0.6985058784484863, "reward_total_composite_std": 0.03456052392721176, "reward_total_mean": 0.6985058784484863, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7361111044883728, "rewards/count_adherence/std": 0.0257172379642725, "rewards/meter/mean": 0.9872919917106628, "rewards/meter/std": 0.03354829549789429, "rewards/repeat_penalty/mean": 0.9615384340286255, "rewards/repeat_penalty/std": 0.029074188321828842, "rewards/total_composite/mean": 0.6985058784484863, "rewards/total_composite/std": 0.03456052392721176, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 3124 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0014961521519580856, "epoch": 0.12551713057798128, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.333333333333335e-07, "loss": 0.0, "num_tokens": 7112433.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0014132261276245, "sampling/importance_sampling_ratio/mean": 1.0001718997955322, "sampling/importance_sampling_ratio/min": 0.9985162019729614, "sampling/sampling_logp_difference/max": 0.0014849056024104357, "sampling/sampling_logp_difference/mean": 0.00018432487559039146, "step": 3125 }, { "clip_ratio/high_max": 0.026507085654884577, "clip_ratio/high_mean": 0.026507085654884577, "clip_ratio/low_mean": 0.007205356378108263, "clip_ratio/low_min": 0.007205356378108263, "clip_ratio/region_mean": 0.03371244203299284, "completions/clipped_ratio": 0.0, "completions/max_length": 423.0, "completions/max_terminated_length": 423.0, "completions/mean_length": 418.375, "completions/mean_terminated_length": 418.375, "completions/min_length": 411.0, "completions/min_terminated_length": 411.0, "entropy": 0.51447868719697, "epoch": 0.12555729605976623, "frac_reward_zero_std": 0.0, "grad_norm": 1.4976996183395386, "learning_rate": 5.303030303030304e-07, "loss": -0.0003, "num_tokens": 7117396.0, "reward": 0.8251492977142334, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8365384340286255, "reward_count_adherence_std": 0.027196412906050682, "reward_meter_mean": 0.9984115362167358, "reward_meter_std": 0.0010889971163123846, "reward_repeat_penalty_mean": 0.988095223903656, "reward_repeat_penalty_std": 0.033671751618385315, "reward_std": 0.036347318440675735, "reward_total_composite_mean": 0.8251492977142334, "reward_total_composite_std": 0.03634733706712723, "reward_total_mean": 0.8251492977142334, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8365384340286255, "rewards/count_adherence/std": 0.027196412906050682, "rewards/meter/mean": 0.9984115362167358, "rewards/meter/std": 0.0010889971163123846, "rewards/repeat_penalty/mean": 0.988095223903656, "rewards/repeat_penalty/std": 0.033671751618385315, "rewards/total_composite/mean": 0.8251492977142334, "rewards/total_composite/std": 0.03634733706712723, "sampling/importance_sampling_ratio/max": 1.753551959991455, "sampling/importance_sampling_ratio/mean": 1.0117367506027222, "sampling/importance_sampling_ratio/min": 0.21122083067893982, "sampling/sampling_logp_difference/max": 1.5548510551452637, "sampling/sampling_logp_difference/mean": 0.05572379007935524, "step": 3126 }, { "clip_ratio/high_max": 0.022230799309909344, "clip_ratio/high_mean": 0.022230799309909344, "clip_ratio/low_mean": 0.00941323209553957, "clip_ratio/low_min": 0.00941323209553957, "clip_ratio/region_mean": 0.031644031405448914, "completions/clipped_ratio": 0.0, "completions/max_length": 468.0, "completions/max_terminated_length": 468.0, "completions/mean_length": 429.75, "completions/mean_terminated_length": 429.75, "completions/min_length": 418.0, "completions/min_terminated_length": 418.0, "entropy": 0.49032482132315636, "epoch": 0.1255974615415512, "frac_reward_zero_std": 0.0, "grad_norm": 1.312803030014038, "learning_rate": 5.272727272727273e-07, "loss": -0.0212, "num_tokens": 7122682.0, "reward": 0.7802447080612183, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7946428060531616, "reward_count_adherence_std": 0.025253823027014732, "reward_meter_mean": 0.9992569088935852, "reward_meter_std": 7.736113184364513e-05, "reward_repeat_penalty_mean": 0.9824134111404419, "reward_repeat_penalty_std": 0.024280980229377747, "reward_std": 0.03580417484045029, "reward_total_composite_mean": 0.7802447080612183, "reward_total_composite_std": 0.035804178565740585, "reward_total_mean": 0.7802447080612183, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7946428060531616, "rewards/count_adherence/std": 0.025253823027014732, "rewards/meter/mean": 0.9992569088935852, "rewards/meter/std": 7.736113184364513e-05, "rewards/repeat_penalty/mean": 0.9824134111404419, "rewards/repeat_penalty/std": 0.024280980229377747, "rewards/total_composite/mean": 0.7802447080612183, "rewards/total_composite/std": 0.035804178565740585, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0106730461120605, "sampling/importance_sampling_ratio/min": 2.6426951080793515e-05, "sampling/sampling_logp_difference/max": 10.541126251220703, "sampling/sampling_logp_difference/mean": 0.05896211043000221, "step": 3127 }, { "clip_ratio/high_max": 0.010851842351257801, "clip_ratio/high_mean": 0.010851842351257801, "clip_ratio/low_mean": 0.010725616884883493, "clip_ratio/low_min": 0.010725616884883493, "clip_ratio/region_mean": 0.021577459236141294, "completions/clipped_ratio": 0.0, "completions/max_length": 189.0, "completions/max_terminated_length": 189.0, "completions/mean_length": 186.0, "completions/mean_terminated_length": 186.0, "completions/min_length": 182.0, "completions/min_terminated_length": 182.0, "entropy": 0.2618657611310482, "epoch": 0.12563762702333614, "frac_reward_zero_std": 0.0, "grad_norm": 1.9481908082962036, "learning_rate": 5.242424242424243e-07, "loss": 0.0112, "num_tokens": 7125754.0, "reward": 0.9295545816421509, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975753426551819, "reward_meter_std": 0.000392713351175189, "reward_repeat_penalty_mean": 0.9318182468414307, "reward_repeat_penalty_std": 0.04208271950483322, "reward_std": 0.04186920449137688, "reward_total_composite_mean": 0.9295545816421509, "reward_total_composite_std": 0.041869208216667175, "reward_total_mean": 0.9295545816421509, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975753426551819, "rewards/meter/std": 0.000392713351175189, "rewards/repeat_penalty/mean": 0.9318182468414307, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.9295545816421509, "rewards/total_composite/std": 0.041869208216667175, "sampling/importance_sampling_ratio/max": 1.669154405593872, "sampling/importance_sampling_ratio/mean": 1.00235915184021, "sampling/importance_sampling_ratio/min": 0.31220078468322754, "sampling/sampling_logp_difference/max": 1.1641087532043457, "sampling/sampling_logp_difference/mean": 0.031066881492733955, "step": 3128 }, { "clip_ratio/high_max": 0.011735557112842798, "clip_ratio/high_mean": 0.011735557112842798, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.011735557112842798, "completions/clipped_ratio": 0.0, "completions/max_length": 130.0, "completions/max_terminated_length": 130.0, "completions/mean_length": 127.375, "completions/mean_terminated_length": 127.375, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.14594370871782303, "epoch": 0.1256777925051211, "frac_reward_zero_std": 0.0, "grad_norm": 2.561624526977539, "learning_rate": 5.212121212121213e-07, "loss": 0.0067, "num_tokens": 7128173.0, "reward": 0.9787326455116272, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9965527057647705, "reward_meter_std": 0.0038294827099889517, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.04999341070652008, "reward_total_composite_mean": 0.9787326455116272, "reward_total_composite_std": 0.04999339208006859, "reward_total_mean": 0.9787326455116272, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9965527057647705, "rewards/meter/std": 0.0038294827099889517, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9787326455116272, "rewards/total_composite/std": 0.04999339208006859, "sampling/importance_sampling_ratio/max": 1.5255719423294067, "sampling/importance_sampling_ratio/mean": 1.0033414363861084, "sampling/importance_sampling_ratio/min": 0.30053746700286865, "sampling/sampling_logp_difference/max": 1.2021827697753906, "sampling/sampling_logp_difference/mean": 0.017186593264341354, "step": 3129 }, { "clip_ratio/high_max": 0.007722355774603784, "clip_ratio/high_mean": 0.007722355774603784, "clip_ratio/low_mean": 0.020387701224535704, "clip_ratio/low_min": 0.020387701224535704, "clip_ratio/region_mean": 0.028110056999139488, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 65.75, "completions/mean_terminated_length": 65.75, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.23753815703094006, "epoch": 0.12571795798690605, "frac_reward_zero_std": 0.0, "grad_norm": 3.1293835639953613, "learning_rate": 5.181818181818182e-07, "loss": 0.0179, "num_tokens": 7129971.0, "reward": 0.9895097613334656, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9895097613334656, "reward_meter_std": 0.006478929426521063, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006478935480117798, "reward_total_composite_mean": 0.9895097613334656, "reward_total_composite_std": 0.006478929426521063, "reward_total_mean": 0.9895097613334656, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9895097613334656, "rewards/meter/std": 0.006478929426521063, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9895097613334656, "rewards/total_composite/std": 0.006478929426521063, "sampling/importance_sampling_ratio/max": 1.572881817817688, "sampling/importance_sampling_ratio/mean": 1.008150577545166, "sampling/importance_sampling_ratio/min": 0.37988272309303284, "sampling/sampling_logp_difference/max": 0.9678926467895508, "sampling/sampling_logp_difference/mean": 0.02998344786465168, "step": 3130 }, { "clip_ratio/high_max": 0.009557109675370157, "clip_ratio/high_mean": 0.009557109675370157, "clip_ratio/low_mean": 0.005769230891019106, "clip_ratio/low_min": 0.005769230891019106, "clip_ratio/region_mean": 0.015326340566389263, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.375, "completions/mean_terminated_length": 65.375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.10212477203458548, "epoch": 0.125758123468691, "frac_reward_zero_std": 0.0, "grad_norm": 3.762568473815918, "learning_rate": 5.151515151515152e-07, "loss": 0.0024, "num_tokens": 7131734.0, "reward": 0.9759130477905273, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9759130477905273, "reward_meter_std": 0.04954683780670166, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.04954684525728226, "reward_total_composite_mean": 0.9759130477905273, "reward_total_composite_std": 0.04954683780670166, "reward_total_mean": 0.9759130477905273, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9759130477905273, "rewards/meter/std": 0.04954683780670166, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9759130477905273, "rewards/total_composite/std": 0.04954683780670166, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0054841041564941, "sampling/importance_sampling_ratio/min": 0.32048511505126953, "sampling/sampling_logp_difference/max": 1.4279863834381104, "sampling/sampling_logp_difference/mean": 0.020955001935362816, "step": 3131 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.005597014795057476, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.05024628387764096, "epoch": 0.12579828895047596, "frac_reward_zero_std": 0.0, "grad_norm": 0.13032668828964233, "learning_rate": 5.121212121212121e-07, "loss": 0.0002, "num_tokens": 7133676.0, "reward": 0.9981508255004883, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981508255004883, "reward_meter_std": 6.344345820252784e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.347765065584099e-06, "reward_total_composite_mean": 0.9981508255004883, "reward_total_composite_std": 6.344345820252784e-06, "reward_total_mean": 0.9981508255004883, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981508255004883, "rewards/meter/std": 6.344345820252784e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981508255004883, "rewards/total_composite/std": 6.344345820252784e-06, "sampling/importance_sampling_ratio/max": 1.516728162765503, "sampling/importance_sampling_ratio/mean": 1.0008666515350342, "sampling/importance_sampling_ratio/min": 0.5628383159637451, "sampling/sampling_logp_difference/max": 0.5747629404067993, "sampling/sampling_logp_difference/mean": 0.007171308156102896, "step": 3132 }, { "clip_ratio/high_max": 0.006225490127690136, "clip_ratio/high_mean": 0.006225490127690136, "clip_ratio/low_mean": 0.00993861141614616, "clip_ratio/low_min": 0.00993861141614616, "clip_ratio/region_mean": 0.016164101543836296, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 100.5, "completions/mean_terminated_length": 100.5, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.17687924578785896, "epoch": 0.1258384544322609, "frac_reward_zero_std": 0.0, "grad_norm": 2.7664103507995605, "learning_rate": 5.090909090909092e-07, "loss": 0.0045, "num_tokens": 7135896.0, "reward": 0.999111533164978, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999111533164978, "reward_meter_std": 0.00020586077880579978, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020586182654369622, "reward_total_composite_mean": 0.999111533164978, "reward_total_composite_std": 0.00020586077880579978, "reward_total_mean": 0.999111533164978, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999111533164978, "rewards/meter/std": 0.00020586077880579978, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999111533164978, "rewards/total_composite/std": 0.00020586077880579978, "sampling/importance_sampling_ratio/max": 1.8379536867141724, "sampling/importance_sampling_ratio/mean": 1.0036976337432861, "sampling/importance_sampling_ratio/min": 0.20234650373458862, "sampling/sampling_logp_difference/max": 1.5977736711502075, "sampling/sampling_logp_difference/mean": 0.025510529056191444, "step": 3133 }, { "clip_ratio/high_max": 0.020418287720531225, "clip_ratio/high_mean": 0.020418287720531225, "clip_ratio/low_mean": 0.01586226187646389, "clip_ratio/low_min": 0.01586226187646389, "clip_ratio/region_mean": 0.036280549596995115, "completions/clipped_ratio": 0.0, "completions/max_length": 112.0, "completions/max_terminated_length": 112.0, "completions/mean_length": 110.125, "completions/mean_terminated_length": 110.125, "completions/min_length": 107.0, "completions/min_terminated_length": 107.0, "entropy": 0.2725033536553383, "epoch": 0.12587861991404586, "frac_reward_zero_std": 0.0, "grad_norm": 5.516884803771973, "learning_rate": 5.060606060606061e-07, "loss": 0.0178, "num_tokens": 7138289.0, "reward": 0.9024706482887268, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9559026956558228, "reward_meter_std": 0.015433711931109428, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.05270349606871605, "reward_total_composite_mean": 0.9024706482887268, "reward_total_composite_std": 0.052703484892845154, "reward_total_mean": 0.9024706482887268, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9559026956558228, "rewards/meter/std": 0.015433711931109428, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.9024706482887268, "rewards/total_composite/std": 0.052703484892845154, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0074241161346436, "sampling/importance_sampling_ratio/min": 0.12714816629886627, "sampling/sampling_logp_difference/max": 2.0624022483825684, "sampling/sampling_logp_difference/mean": 0.05570008605718613, "step": 3134 }, { "clip_ratio/high_max": 0.006212871172465384, "clip_ratio/high_mean": 0.006212871172465384, "clip_ratio/low_mean": 0.011116024106740952, "clip_ratio/low_min": 0.011116024106740952, "clip_ratio/region_mean": 0.017328895279206336, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 100.75, "completions/mean_terminated_length": 100.75, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.16831262316554785, "epoch": 0.12591878539583082, "frac_reward_zero_std": 0.0, "grad_norm": 1.680987000465393, "learning_rate": 5.03030303030303e-07, "loss": 0.0022, "num_tokens": 7140487.0, "reward": 0.9992322325706482, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992322325706482, "reward_meter_std": 0.00010666978050721809, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010666977323126048, "reward_total_composite_mean": 0.9992322325706482, "reward_total_composite_std": 0.00010666978050721809, "reward_total_mean": 0.9992322325706482, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992322325706482, "rewards/meter/std": 0.00010666978050721809, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992322325706482, "rewards/total_composite/std": 0.00010666978050721809, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.006977915763855, "sampling/importance_sampling_ratio/min": 0.25051194429397583, "sampling/sampling_logp_difference/max": 1.3842487335205078, "sampling/sampling_logp_difference/mean": 0.029712356626987457, "step": 3135 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0006697150638501626, "epoch": 0.12595895087761577, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 5.000000000000001e-07, "loss": 0.0, "num_tokens": 7142015.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0015604496002197, "sampling/importance_sampling_ratio/mean": 1.0000699758529663, "sampling/importance_sampling_ratio/min": 0.999688982963562, "sampling/sampling_logp_difference/max": 0.0015591848641633987, "sampling/sampling_logp_difference/mean": 7.456184539478272e-05, "step": 3136 }, { "clip_ratio/high_max": 0.011029411805793643, "clip_ratio/high_mean": 0.011029411805793643, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.014705882407724857, "completions/clipped_ratio": 0.0, "completions/max_length": 68.0, "completions/max_terminated_length": 68.0, "completions/mean_length": 68.0, "completions/mean_terminated_length": 68.0, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.058580921962857246, "epoch": 0.12599911635940073, "frac_reward_zero_std": 0.0, "grad_norm": 1.265021800994873, "learning_rate": 4.96969696969697e-07, "loss": 0.0016, "num_tokens": 7143751.0, "reward": 0.9994433522224426, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994433522224426, "reward_meter_std": 8.680017344886437e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.68034694576636e-05, "reward_total_composite_mean": 0.9994433522224426, "reward_total_composite_std": 8.680017344886437e-05, "reward_total_mean": 0.9994433522224426, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994433522224426, "rewards/meter/std": 8.680017344886437e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994433522224426, "rewards/total_composite/std": 8.680017344886437e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0005712509155273, "sampling/importance_sampling_ratio/min": 0.4000520408153534, "sampling/sampling_logp_difference/max": 1.1407603025436401, "sampling/sampling_logp_difference/mean": 0.01834726519882679, "step": 3137 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0030487803742289543, "clip_ratio/low_min": 0.0030487803742289543, "clip_ratio/region_mean": 0.0030487803742289543, "completions/clipped_ratio": 0.0, "completions/max_length": 42.0, "completions/max_terminated_length": 42.0, "completions/mean_length": 40.875, "completions/mean_terminated_length": 40.875, "completions/min_length": 40.0, "completions/min_terminated_length": 40.0, "entropy": 0.12436437420547009, "epoch": 0.12603928184118568, "frac_reward_zero_std": 0.0, "grad_norm": 2.8887453079223633, "learning_rate": 4.93939393939394e-07, "loss": -0.0077, "num_tokens": 7145214.0, "reward": 0.9984819889068604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9984819889068604, "reward_meter_std": 0.0004429934488143772, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00044299106230027974, "reward_total_composite_mean": 0.9984819889068604, "reward_total_composite_std": 0.0004429934488143772, "reward_total_mean": 0.9984819889068604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9984819889068604, "rewards/meter/std": 0.0004429934488143772, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9984819889068604, "rewards/total_composite/std": 0.0004429934488143772, "sampling/importance_sampling_ratio/max": 1.4185824394226074, "sampling/importance_sampling_ratio/mean": 1.0034947395324707, "sampling/importance_sampling_ratio/min": 0.6600699424743652, "sampling/sampling_logp_difference/max": 0.41540956497192383, "sampling/sampling_logp_difference/mean": 0.012076342478394508, "step": 3138 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.005597014795057476, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.04900990845635533, "epoch": 0.12607944732297063, "frac_reward_zero_std": 0.0, "grad_norm": 0.1556718647480011, "learning_rate": 4.909090909090909e-07, "loss": -0.0001, "num_tokens": 7146949.0, "reward": 0.9981504678726196, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981504678726196, "reward_meter_std": 9.697491805127356e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 9.708711331768427e-06, "reward_total_composite_mean": 0.9981504678726196, "reward_total_composite_std": 9.697491805127356e-06, "reward_total_mean": 0.9981504678726196, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981504678726196, "rewards/meter/std": 9.697491805127356e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981504678726196, "rewards/total_composite/std": 9.697491805127356e-06, "sampling/importance_sampling_ratio/max": 1.3643027544021606, "sampling/importance_sampling_ratio/mean": 1.0014523267745972, "sampling/importance_sampling_ratio/min": 0.5597096681594849, "sampling/sampling_logp_difference/max": 0.5803370475769043, "sampling/sampling_logp_difference/mean": 0.007338312920182943, "step": 3139 }, { "clip_ratio/high_max": 0.02794860501307994, "clip_ratio/high_mean": 0.02794860501307994, "clip_ratio/low_mean": 0.005737704690545797, "clip_ratio/low_min": 0.005737704690545797, "clip_ratio/region_mean": 0.03368630970362574, "completions/clipped_ratio": 0.0, "completions/max_length": 322.0, "completions/max_terminated_length": 322.0, "completions/mean_length": 311.625, "completions/mean_terminated_length": 311.625, "completions/min_length": 305.0, "completions/min_terminated_length": 305.0, "entropy": 0.447547797113657, "epoch": 0.1261196128047556, "frac_reward_zero_std": 0.0, "grad_norm": 2.1409451961517334, "learning_rate": 4.878787878787879e-07, "loss": -0.0074, "num_tokens": 7151106.0, "reward": 0.9965137243270874, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9965137243270874, "reward_meter_std": 0.00763244554400444, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0076324427500367165, "reward_total_composite_mean": 0.9965137243270874, "reward_total_composite_std": 0.00763244554400444, "reward_total_mean": 0.9965137243270874, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9965137243270874, "rewards/meter/std": 0.00763244554400444, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9965137243270874, "rewards/total_composite/std": 0.00763244554400444, "sampling/importance_sampling_ratio/max": 1.9414808750152588, "sampling/importance_sampling_ratio/mean": 1.0111055374145508, "sampling/importance_sampling_ratio/min": 0.2587974965572357, "sampling/sampling_logp_difference/max": 1.3517093658447266, "sampling/sampling_logp_difference/mean": 0.0488644503057003, "step": 3140 }, { "clip_ratio/high_max": 0.0036764706019312143, "clip_ratio/high_mean": 0.0036764706019312143, "clip_ratio/low_mean": 0.0078125, "clip_ratio/low_min": 0.0078125, "clip_ratio/region_mean": 0.011488970601931214, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 33.75, "completions/mean_terminated_length": 33.75, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.10679818876087666, "epoch": 0.12615977828654054, "frac_reward_zero_std": 0.0, "grad_norm": 5.089842319488525, "learning_rate": 4.848484848484849e-07, "loss": -0.0163, "num_tokens": 7152584.0, "reward": 0.9182018041610718, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9182018041610718, "reward_meter_std": 0.2099631130695343, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.2099630981683731, "reward_total_composite_mean": 0.9182018041610718, "reward_total_composite_std": 0.2099631130695343, "reward_total_mean": 0.9182018041610718, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9182018041610718, "rewards/meter/std": 0.2099631130695343, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9182018041610718, "rewards/total_composite/std": 0.2099631130695343, "sampling/importance_sampling_ratio/max": 1.7678229808807373, "sampling/importance_sampling_ratio/mean": 1.0063190460205078, "sampling/importance_sampling_ratio/min": 0.41221490502357483, "sampling/sampling_logp_difference/max": 0.8862104415893555, "sampling/sampling_logp_difference/mean": 0.012079699896275997, "step": 3141 }, { "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/low_mean": 0.0036764706019312143, "clip_ratio/low_min": 0.0036764706019312143, "clip_ratio/region_mean": 0.007197597296908498, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 70.625, "completions/mean_terminated_length": 70.625, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.09765590541064739, "epoch": 0.1261999437683255, "frac_reward_zero_std": 0.0, "grad_norm": 3.1886298656463623, "learning_rate": 4.818181818181818e-07, "loss": -0.0134, "num_tokens": 7154301.0, "reward": 0.9989311099052429, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989311099052429, "reward_meter_std": 0.0014027439756318927, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0014027438592165709, "reward_total_composite_mean": 0.9989311099052429, "reward_total_composite_std": 0.0014027439756318927, "reward_total_mean": 0.9989311099052429, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989311099052429, "rewards/meter/std": 0.0014027439756318927, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989311099052429, "rewards/total_composite/std": 0.0014027439756318927, "sampling/importance_sampling_ratio/max": 1.341347575187683, "sampling/importance_sampling_ratio/mean": 1.0028244256973267, "sampling/importance_sampling_ratio/min": 0.3958625793457031, "sampling/sampling_logp_difference/max": 0.9266881942749023, "sampling/sampling_logp_difference/mean": 0.013442294672131538, "step": 3142 }, { "clip_ratio/high_max": 0.01591338007710874, "clip_ratio/high_mean": 0.01591338007710874, "clip_ratio/low_mean": 0.013188510201871395, "clip_ratio/low_min": 0.013188510201871395, "clip_ratio/region_mean": 0.029101890278980136, "completions/clipped_ratio": 0.0, "completions/max_length": 135.0, "completions/max_terminated_length": 135.0, "completions/mean_length": 133.375, "completions/mean_terminated_length": 133.375, "completions/min_length": 128.0, "completions/min_terminated_length": 128.0, "entropy": 0.29481677152216434, "epoch": 0.12624010925011045, "frac_reward_zero_std": 0.0, "grad_norm": 1.7062424421310425, "learning_rate": 4.787878787878789e-07, "loss": 0.0034, "num_tokens": 7156840.0, "reward": 0.9990321397781372, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990321397781372, "reward_meter_std": 0.00018018556875176728, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00018020268180407584, "reward_total_composite_mean": 0.9990321397781372, "reward_total_composite_std": 0.00018018556875176728, "reward_total_mean": 0.9990321397781372, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990321397781372, "rewards/meter/std": 0.00018018556875176728, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990321397781372, "rewards/total_composite/std": 0.00018018556875176728, "sampling/importance_sampling_ratio/max": 1.9823683500289917, "sampling/importance_sampling_ratio/mean": 1.0070942640304565, "sampling/importance_sampling_ratio/min": 0.3141377568244934, "sampling/sampling_logp_difference/max": 1.157923698425293, "sampling/sampling_logp_difference/mean": 0.035543572157621384, "step": 3143 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0032840893836691976, "epoch": 0.1262802747318954, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.757575757575758e-07, "loss": 0.0, "num_tokens": 7158856.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0129438638687134, "sampling/importance_sampling_ratio/mean": 1.0003576278686523, "sampling/importance_sampling_ratio/min": 0.9996905326843262, "sampling/sampling_logp_difference/max": 0.012860830873250961, "sampling/sampling_logp_difference/mean": 0.00035932144965045154, "step": 3144 }, { "clip_ratio/high_max": 0.014405153575353324, "clip_ratio/high_mean": 0.014405153575353324, "clip_ratio/low_mean": 0.014030612306669354, "clip_ratio/low_min": 0.014030612306669354, "clip_ratio/region_mean": 0.02843576588202268, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 96.5, "completions/mean_terminated_length": 96.5, "completions/min_length": 95.0, "completions/min_terminated_length": 95.0, "entropy": 0.23462153412401676, "epoch": 0.12632044021368036, "frac_reward_zero_std": 0.0, "grad_norm": 3.3756773471832275, "learning_rate": 4.7272727272727273e-07, "loss": 0.0038, "num_tokens": 7161044.0, "reward": 0.9917601346969604, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9917601346969604, "reward_meter_std": 0.004016265273094177, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004016268998384476, "reward_total_composite_mean": 0.9917601346969604, "reward_total_composite_std": 0.004016265273094177, "reward_total_mean": 0.9917601346969604, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9917601346969604, "rewards/meter/std": 0.004016265273094177, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9917601346969604, "rewards/total_composite/std": 0.004016265273094177, "sampling/importance_sampling_ratio/max": 1.6533253192901611, "sampling/importance_sampling_ratio/mean": 1.0063228607177734, "sampling/importance_sampling_ratio/min": 0.3019343912601471, "sampling/sampling_logp_difference/max": 1.1975455284118652, "sampling/sampling_logp_difference/mean": 0.031790200620889664, "step": 3145 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.00371554572484456, "epoch": 0.1263606056954653, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.696969696969697e-07, "loss": 0.0, "num_tokens": 7162452.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0081679821014404, "sampling/importance_sampling_ratio/mean": 1.0002386569976807, "sampling/importance_sampling_ratio/min": 0.9950645565986633, "sampling/sampling_logp_difference/max": 0.008134791627526283, "sampling/sampling_logp_difference/mean": 0.00029772642301395535, "step": 3146 }, { "clip_ratio/high_max": 0.017632243572734296, "clip_ratio/high_mean": 0.017632243572734296, "clip_ratio/low_mean": 0.004168422543443739, "clip_ratio/low_min": 0.004168422543443739, "clip_ratio/region_mean": 0.021800666116178036, "completions/clipped_ratio": 0.0, "completions/max_length": 339.0, "completions/max_terminated_length": 339.0, "completions/mean_length": 315.625, "completions/mean_terminated_length": 315.625, "completions/min_length": 309.0, "completions/min_terminated_length": 309.0, "entropy": 0.41412847116589546, "epoch": 0.12640077117725027, "frac_reward_zero_std": 0.0, "grad_norm": 1.647410273551941, "learning_rate": 4.666666666666667e-07, "loss": 0.0288, "num_tokens": 7166497.0, "reward": 0.9686882495880127, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.984375, "reward_count_adherence_std": 0.04419417306780815, "reward_meter_mean": 0.9990444779396057, "reward_meter_std": 0.00020862040400970727, "reward_repeat_penalty_mean": 0.9843137264251709, "reward_repeat_penalty_std": 0.029120875522494316, "reward_std": 0.06351151317358017, "reward_total_composite_mean": 0.9686882495880127, "reward_total_composite_std": 0.06351150572299957, "reward_total_mean": 0.9686882495880127, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.984375, "rewards/count_adherence/std": 0.04419417306780815, "rewards/meter/mean": 0.9990444779396057, "rewards/meter/std": 0.00020862040400970727, "rewards/repeat_penalty/mean": 0.9843137264251709, "rewards/repeat_penalty/std": 0.029120875522494316, "rewards/total_composite/mean": 0.9686882495880127, "rewards/total_composite/std": 0.06351150572299957, "sampling/importance_sampling_ratio/max": 1.7542108297348022, "sampling/importance_sampling_ratio/mean": 1.0113532543182373, "sampling/importance_sampling_ratio/min": 0.25884810090065, "sampling/sampling_logp_difference/max": 1.3515138626098633, "sampling/sampling_logp_difference/mean": 0.04411856085062027, "step": 3147 }, { "clip_ratio/high_max": 0.00821063295006752, "clip_ratio/high_mean": 0.00821063295006752, "clip_ratio/low_mean": 0.008188590873032808, "clip_ratio/low_min": 0.008188590873032808, "clip_ratio/region_mean": 0.01639922382310033, "completions/clipped_ratio": 0.0, "completions/max_length": 107.0, "completions/max_terminated_length": 107.0, "completions/mean_length": 106.75, "completions/mean_terminated_length": 106.75, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.14908354356884956, "epoch": 0.12644093665903522, "frac_reward_zero_std": 0.0, "grad_norm": 0.7614604234695435, "learning_rate": 4.6363636363636365e-07, "loss": 0.0009, "num_tokens": 7168719.0, "reward": 0.9992802143096924, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992802143096924, "reward_meter_std": 0.0001533372706035152, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00015333489864133298, "reward_total_composite_mean": 0.9992802143096924, "reward_total_composite_std": 0.0001533372706035152, "reward_total_mean": 0.9992802143096924, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992802143096924, "rewards/meter/std": 0.0001533372706035152, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992802143096924, "rewards/total_composite/std": 0.0001533372706035152, "sampling/importance_sampling_ratio/max": 1.5043672323226929, "sampling/importance_sampling_ratio/mean": 1.0032622814178467, "sampling/importance_sampling_ratio/min": 0.30748796463012695, "sampling/sampling_logp_difference/max": 1.1793193817138672, "sampling/sampling_logp_difference/mean": 0.013726499862968922, "step": 3148 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.12648110214082017, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 4.6060606060606064e-07, "loss": 0.0, "num_tokens": 7170575.0, "reward": 0.6074195504188538, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.6687499284744263, "reward_count_adherence_std": 0.025877466425299644, "reward_meter_mean": 0.9272494316101074, "reward_meter_std": 0.10411977022886276, "reward_repeat_penalty_mean": 0.9813033938407898, "reward_repeat_penalty_std": 0.03969252109527588, "reward_std": 0.06437568366527557, "reward_total_composite_mean": 0.6074195504188538, "reward_total_composite_std": 0.06437568366527557, "reward_total_mean": 0.6074195504188538, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.6687499284744263, "rewards/count_adherence/std": 0.025877466425299644, "rewards/meter/mean": 0.9272494316101074, "rewards/meter/std": 0.10411977022886276, "rewards/repeat_penalty/mean": 0.9813033938407898, "rewards/repeat_penalty/std": 0.03969252109527588, "rewards/total_composite/mean": 0.6074195504188538, "rewards/total_composite/std": 0.06437568366527557, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 3149 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0020491802133619785, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.021077871089801192, "epoch": 0.12652126762260513, "frac_reward_zero_std": 0.0, "grad_norm": 0.012665612623095512, "learning_rate": 4.5757575757575764e-07, "loss": 0.0003, "num_tokens": 7172439.0, "reward": 0.997339129447937, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.997339129447937, "reward_meter_std": 3.875939285080676e-07, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.8628223819614504e-07, "reward_total_composite_mean": 0.997339129447937, "reward_total_composite_std": 3.875939285080676e-07, "reward_total_mean": 0.997339129447937, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.997339129447937, "rewards/meter/std": 3.875939285080676e-07, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.997339129447937, "rewards/total_composite/std": 3.875939285080676e-07, "sampling/importance_sampling_ratio/max": 1.506388783454895, "sampling/importance_sampling_ratio/mean": 0.9999790787696838, "sampling/importance_sampling_ratio/min": 0.4620819389820099, "sampling/sampling_logp_difference/max": 0.7720130681991577, "sampling/sampling_logp_difference/mean": 0.004391301888972521, "step": 3150 }, { "epoch": 0.12652126762260513, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.038461538461538464, "eval_completions/max_length": 436.2307692307692, "eval_completions/max_terminated_length": 401.84615384615387, "eval_completions/mean_length": 217.70192307692307, "eval_completions/mean_terminated_length": 206.14835533728967, "eval_completions/min_length": 61.69230769230769, "eval_completions/min_terminated_length": 61.69230769230769, "eval_entropy": 0.37681426910253674, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 7172439.0, "eval_reward": 0.7355112204184899, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.963410904774299, "eval_reward_count_adherence_std": 0.054318822060640044, "eval_reward_meter_mean": 0.7973346756054804, "eval_reward_meter_std": 0.3138508295210508, "eval_reward_repeat_penalty_mean": 0.9485706686973572, "eval_reward_repeat_penalty_std": 0.08086307346820831, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7355112204184899, "eval_reward_total_composite_std": 0.31246042595459866, "eval_reward_total_mean": 0.7355112204184899, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.963410904774299, "eval_rewards/count_adherence/std": 0.054318822060640044, "eval_rewards/meter/mean": 0.7973346756054804, "eval_rewards/meter/std": 0.3138508295210508, "eval_rewards/repeat_penalty/mean": 0.9485706686973572, "eval_rewards/repeat_penalty/std": 0.08086307346820831, "eval_rewards/total_composite/mean": 0.7355112204184899, "eval_rewards/total_composite/std": 0.31246042595459866, "eval_runtime": 80.4009, "eval_samples_per_second": 1.294, "eval_sampling/importance_sampling_ratio/max": 1.4886829761358409, "eval_sampling/importance_sampling_ratio/mean": 1.0091725221047034, "eval_sampling/importance_sampling_ratio/min": 0.33571984332341415, "eval_sampling/sampling_logp_difference/max": 1.1065230736365685, "eval_sampling/sampling_logp_difference/mean": 0.0324688284442975, "eval_steps_per_second": 0.162, "step": 3150 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0016731252981116995, "epoch": 0.12656143310439008, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.5454545454545457e-07, "loss": 0.0, "num_tokens": 7174111.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0024826526641846, "sampling/importance_sampling_ratio/mean": 1.000199317932129, "sampling/importance_sampling_ratio/min": 0.9986385703086853, "sampling/sampling_logp_difference/max": 0.002479560673236847, "sampling/sampling_logp_difference/mean": 0.00020680490706581622, "step": 3151 }, { "clip_ratio/high_max": 0.003846153849735856, "clip_ratio/high_mean": 0.003846153849735856, "clip_ratio/low_mean": 0.007634032750502229, "clip_ratio/low_min": 0.007634032750502229, "clip_ratio/region_mean": 0.011480186600238085, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.125, "completions/mean_terminated_length": 65.125, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.06057531526312232, "epoch": 0.12660159858617503, "frac_reward_zero_std": 0.0, "grad_norm": 1.8619328737258911, "learning_rate": 4.5151515151515156e-07, "loss": 0.0014, "num_tokens": 7175864.0, "reward": 0.9934186935424805, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9934186935424805, "reward_meter_std": 0.00011813976016128436, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011814001481980085, "reward_total_composite_mean": 0.9934186935424805, "reward_total_composite_std": 0.00011813976016128436, "reward_total_mean": 0.9934186935424805, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9934186935424805, "rewards/meter/std": 0.00011813976016128436, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9934186935424805, "rewards/total_composite/std": 0.00011813976016128436, "sampling/importance_sampling_ratio/max": 1.3604105710983276, "sampling/importance_sampling_ratio/mean": 1.0027790069580078, "sampling/importance_sampling_ratio/min": 0.683942973613739, "sampling/sampling_logp_difference/max": 0.37988075613975525, "sampling/sampling_logp_difference/mean": 0.010398104786872864, "step": 3152 }, { "clip_ratio/high_max": 0.006870172452181578, "clip_ratio/high_mean": 0.006870172452181578, "clip_ratio/low_mean": 0.0037722690613009036, "clip_ratio/low_min": 0.0037722690613009036, "clip_ratio/region_mean": 0.010642441513482481, "completions/clipped_ratio": 0.0, "completions/max_length": 204.0, "completions/max_terminated_length": 204.0, "completions/mean_length": 199.375, "completions/mean_terminated_length": 199.375, "completions/min_length": 197.0, "completions/min_terminated_length": 197.0, "entropy": 0.18566425889730453, "epoch": 0.12664176406796, "frac_reward_zero_std": 0.0, "grad_norm": 2.150918483734131, "learning_rate": 4.484848484848485e-07, "loss": 0.0065, "num_tokens": 7179187.0, "reward": 0.9311637878417969, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993048906326294, "reward_meter_std": 0.0001919995847856626, "reward_repeat_penalty_mean": 0.9318181276321411, "reward_repeat_penalty_std": 0.08058229833841324, "reward_std": 0.08044035732746124, "reward_total_composite_mean": 0.9311637878417969, "reward_total_composite_std": 0.08044037967920303, "reward_total_mean": 0.9311637878417969, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993048906326294, "rewards/meter/std": 0.0001919995847856626, "rewards/repeat_penalty/mean": 0.9318181276321411, "rewards/repeat_penalty/std": 0.08058229833841324, "rewards/total_composite/mean": 0.9311637878417969, "rewards/total_composite/std": 0.08044037967920303, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0063588619232178, "sampling/importance_sampling_ratio/min": 0.31784144043922424, "sampling/sampling_logp_difference/max": 1.146202564239502, "sampling/sampling_logp_difference/mean": 0.020360471680760384, "step": 3153 }, { "clip_ratio/high_max": 0.021670067915692925, "clip_ratio/high_mean": 0.021670067915692925, "clip_ratio/low_mean": 0.007282239850610495, "clip_ratio/low_min": 0.007282239850610495, "clip_ratio/region_mean": 0.02895230776630342, "completions/clipped_ratio": 0.0, "completions/max_length": 105.0, "completions/max_terminated_length": 105.0, "completions/mean_length": 103.625, "completions/mean_terminated_length": 103.625, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.30003978684544563, "epoch": 0.12668192954974494, "frac_reward_zero_std": 0.0, "grad_norm": 3.9876134395599365, "learning_rate": 4.454545454545455e-07, "loss": 0.0047, "num_tokens": 7181304.0, "reward": 0.9898449182510376, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9898449182510376, "reward_meter_std": 0.012184920720756054, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.01218490768224001, "reward_total_composite_mean": 0.9898449182510376, "reward_total_composite_std": 0.012184920720756054, "reward_total_mean": 0.9898449182510376, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9898449182510376, "rewards/meter/std": 0.012184920720756054, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9898449182510376, "rewards/total_composite/std": 0.012184920720756054, "sampling/importance_sampling_ratio/max": 1.9000133275985718, "sampling/importance_sampling_ratio/mean": 1.005436658859253, "sampling/importance_sampling_ratio/min": 0.20812343060970306, "sampling/sampling_logp_difference/max": 1.5696239471435547, "sampling/sampling_logp_difference/mean": 0.036951709538698196, "step": 3154 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.013011510367505252, "epoch": 0.1267220950315299, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.424242424242425e-07, "loss": 0.0, "num_tokens": 7183088.0, "reward": 0.9973388910293579, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973388910293579, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9973388910293579, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9973388910293579, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973388910293579, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973388910293579, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.055625557899475, "sampling/importance_sampling_ratio/mean": 1.0003093481063843, "sampling/importance_sampling_ratio/min": 0.8605263233184814, "sampling/sampling_logp_difference/max": 0.15021102130413055, "sampling/sampling_logp_difference/mean": 0.0017873203614726663, "step": 3155 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0018204424268333241, "epoch": 0.12676226051331485, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.393939393939394e-07, "loss": 0.0, "num_tokens": 7184864.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0035988092422485, "sampling/importance_sampling_ratio/mean": 1.000244379043579, "sampling/importance_sampling_ratio/min": 0.9997955560684204, "sampling/sampling_logp_difference/max": 0.0035922862589359283, "sampling/sampling_logp_difference/mean": 0.0002461693366058171, "step": 3156 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0016821113676996902, "epoch": 0.1268024259950998, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.363636363636364e-07, "loss": 0.0, "num_tokens": 7186584.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0021194219589233, "sampling/importance_sampling_ratio/mean": 1.0001963376998901, "sampling/importance_sampling_ratio/min": 0.9994971752166748, "sampling/sampling_logp_difference/max": 0.002117213560268283, "sampling/sampling_logp_difference/mean": 0.00020154265803284943, "step": 3157 }, { "clip_ratio/high_max": 0.009170901728793979, "clip_ratio/high_mean": 0.009170901728793979, "clip_ratio/low_mean": 0.0035211266949772835, "clip_ratio/low_min": 0.0035211266949772835, "clip_ratio/region_mean": 0.012692028423771262, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 68.75, "completions/mean_terminated_length": 68.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.23758242093026638, "epoch": 0.12684259147688476, "frac_reward_zero_std": 0.0, "grad_norm": 4.2719950675964355, "learning_rate": 4.333333333333334e-07, "loss": 0.0133, "num_tokens": 7188342.0, "reward": 0.9841428399085999, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9841428399085999, "reward_meter_std": 0.03312673419713974, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.03312673047184944, "reward_total_composite_mean": 0.9841428399085999, "reward_total_composite_std": 0.03312673419713974, "reward_total_mean": 0.9841428399085999, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9841428399085999, "rewards/meter/std": 0.03312673419713974, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9841428399085999, "rewards/total_composite/std": 0.03312673419713974, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007124900817871, "sampling/importance_sampling_ratio/min": 0.337809294462204, "sampling/sampling_logp_difference/max": 1.0852737426757812, "sampling/sampling_logp_difference/mean": 0.03139575570821762, "step": 3158 }, { "clip_ratio/high_max": 0.04191699391230941, "clip_ratio/high_mean": 0.04191699391230941, "clip_ratio/low_mean": 0.01434953324496746, "clip_ratio/low_min": 0.01434953324496746, "clip_ratio/region_mean": 0.05626652715727687, "completions/clipped_ratio": 0.0, "completions/max_length": 89.0, "completions/max_terminated_length": 89.0, "completions/mean_length": 86.5, "completions/mean_terminated_length": 86.5, "completions/min_length": 83.0, "completions/min_terminated_length": 83.0, "entropy": 0.24941212497651577, "epoch": 0.1268827569586697, "frac_reward_zero_std": 0.0, "grad_norm": 6.7398481369018555, "learning_rate": 4.3030303030303034e-07, "loss": 0.0047, "num_tokens": 7190330.0, "reward": 0.8583756685256958, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8662310242652893, "reward_meter_std": 0.18711434304714203, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.20775045454502106, "reward_total_composite_mean": 0.8583756685256958, "reward_total_composite_std": 0.20775045454502106, "reward_total_mean": 0.8583756685256958, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8662310242652893, "rewards/meter/std": 0.18711434304714203, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.8583756685256958, "rewards/total_composite/std": 0.20775045454502106, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0018366575241089, "sampling/importance_sampling_ratio/min": 0.3277885615825653, "sampling/sampling_logp_difference/max": 1.4105424880981445, "sampling/sampling_logp_difference/mean": 0.05684678256511688, "step": 3159 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.12692292244045467, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 4.272727272727273e-07, "loss": 0.0, "num_tokens": 7192010.0, "reward": 0.5677040219306946, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.6624999642372131, "reward_count_adherence_std": 0.023145509883761406, "reward_meter_mean": 0.9984534978866577, "reward_meter_std": 0.0011360765201970935, "reward_repeat_penalty_mean": 0.9809472560882568, "reward_repeat_penalty_std": 0.020373545587062836, "reward_std": 0.230849489569664, "reward_total_composite_mean": 0.5677040219306946, "reward_total_composite_std": 0.2308495044708252, "reward_total_mean": 0.5677040219306946, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.6624999642372131, "rewards/count_adherence/std": 0.023145509883761406, "rewards/meter/mean": 0.9984534978866577, "rewards/meter/std": 0.0011360765201970935, "rewards/repeat_penalty/mean": 0.9809472560882568, "rewards/repeat_penalty/std": 0.020373545587062836, "rewards/total_composite/mean": 0.5677040219306946, "rewards/total_composite/std": 0.2308495044708252, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 3160 }, { "clip_ratio/high_max": 0.013319394318386912, "clip_ratio/high_mean": 0.013319394318386912, "clip_ratio/low_mean": 0.008485189639031887, "clip_ratio/low_min": 0.008485189639031887, "clip_ratio/region_mean": 0.0218045839574188, "completions/clipped_ratio": 0.0, "completions/max_length": 359.0, "completions/max_terminated_length": 359.0, "completions/mean_length": 349.625, "completions/mean_terminated_length": 349.625, "completions/min_length": 338.0, "completions/min_terminated_length": 338.0, "entropy": 0.35866962373256683, "epoch": 0.12696308792223962, "frac_reward_zero_std": 0.0, "grad_norm": 1.4183757305145264, "learning_rate": 4.242424242424243e-07, "loss": -0.0163, "num_tokens": 7196839.0, "reward": 0.7685060501098633, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8020833134651184, "reward_count_adherence_std": 0.0431290864944458, "reward_meter_mean": 0.9987608194351196, "reward_meter_std": 0.00043402795563451946, "reward_repeat_penalty_mean": 0.9601608514785767, "reward_repeat_penalty_std": 0.02460995875298977, "reward_std": 0.030336128547787666, "reward_total_composite_mean": 0.7685060501098633, "reward_total_composite_std": 0.03033612295985222, "reward_total_mean": 0.7685060501098633, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8020833134651184, "rewards/count_adherence/std": 0.0431290864944458, "rewards/meter/mean": 0.9987608194351196, "rewards/meter/std": 0.00043402795563451946, "rewards/repeat_penalty/mean": 0.9601608514785767, "rewards/repeat_penalty/std": 0.02460995875298977, "rewards/total_composite/mean": 0.7685060501098633, "rewards/total_composite/std": 0.03033612295985222, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0060912370681763, "sampling/importance_sampling_ratio/min": 0.1878807246685028, "sampling/sampling_logp_difference/max": 1.671947956085205, "sampling/sampling_logp_difference/mean": 0.034961603581905365, "step": 3161 }, { "clip_ratio/high_max": 0.0035211266949772835, "clip_ratio/high_mean": 0.0035211266949772835, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0035211266949772835, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.125, "completions/mean_terminated_length": 71.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.07914929743856192, "epoch": 0.12700325340402457, "frac_reward_zero_std": 0.0, "grad_norm": 0.1628890037536621, "learning_rate": 4.2121212121212126e-07, "loss": -0.0002, "num_tokens": 7198784.0, "reward": 0.9994406700134277, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994406700134277, "reward_meter_std": 1.3193358427088242e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.3199281056586187e-05, "reward_total_composite_mean": 0.9994406700134277, "reward_total_composite_std": 1.3193358427088242e-05, "reward_total_mean": 0.9994406700134277, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994406700134277, "rewards/meter/std": 1.3193358427088242e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994406700134277, "rewards/total_composite/std": 1.3193358427088242e-05, "sampling/importance_sampling_ratio/max": 1.3171619176864624, "sampling/importance_sampling_ratio/mean": 1.003661036491394, "sampling/importance_sampling_ratio/min": 0.716238796710968, "sampling/sampling_logp_difference/max": 0.3337416648864746, "sampling/sampling_logp_difference/mean": 0.0068392762914299965, "step": 3162 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.003859572723740712, "epoch": 0.12704341888580953, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.181818181818182e-07, "loss": 0.0, "num_tokens": 7200688.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0227203369140625, "sampling/importance_sampling_ratio/mean": 1.0004879236221313, "sampling/importance_sampling_ratio/min": 0.9993630647659302, "sampling/sampling_logp_difference/max": 0.022466091439127922, "sampling/sampling_logp_difference/mean": 0.0004915767349302769, "step": 3163 }, { "clip_ratio/high_max": 0.019021739484742284, "clip_ratio/high_mean": 0.019021739484742284, "clip_ratio/low_mean": 0.0027173913549631834, "clip_ratio/low_min": 0.0027173913549631834, "clip_ratio/region_mean": 0.021739130839705467, "completions/clipped_ratio": 0.0, "completions/max_length": 46.0, "completions/max_terminated_length": 46.0, "completions/mean_length": 45.875, "completions/mean_terminated_length": 45.875, "completions/min_length": 45.0, "completions/min_terminated_length": 45.0, "entropy": 0.10774505604058504, "epoch": 0.12708358436759448, "frac_reward_zero_std": 0.0, "grad_norm": 7.426182270050049, "learning_rate": 4.1515151515151513e-07, "loss": -0.0002, "num_tokens": 7202591.0, "reward": 0.9413228631019592, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9413228631019592, "reward_meter_std": 0.004475572612136602, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.004475583788007498, "reward_total_composite_mean": 0.9413228631019592, "reward_total_composite_std": 0.004475572612136602, "reward_total_mean": 0.9413228631019592, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9413228631019592, "rewards/meter/std": 0.004475572612136602, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9413228631019592, "rewards/total_composite/std": 0.004475572612136602, "sampling/importance_sampling_ratio/max": 1.8759413957595825, "sampling/importance_sampling_ratio/mean": 0.9981755614280701, "sampling/importance_sampling_ratio/min": 0.10473800450563431, "sampling/sampling_logp_difference/max": 2.256293296813965, "sampling/sampling_logp_difference/mean": 0.033416956663131714, "step": 3164 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0008098809266812168, "epoch": 0.12712374984937944, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.121212121212122e-07, "loss": 0.0, "num_tokens": 7204447.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0019134283065796, "sampling/importance_sampling_ratio/mean": 1.000064730644226, "sampling/importance_sampling_ratio/min": 0.9977367520332336, "sampling/sampling_logp_difference/max": 0.0022658593952655792, "sampling/sampling_logp_difference/mean": 0.00010137181379832327, "step": 3165 }, { "clip_ratio/high_max": 0.0018656715983524919, "clip_ratio/high_mean": 0.0018656715983524919, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.0037313431967049837, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.052269697189331055, "epoch": 0.1271639153311644, "frac_reward_zero_std": 0.0, "grad_norm": 0.4401102066040039, "learning_rate": 4.090909090909091e-07, "loss": -0.0001, "num_tokens": 7206119.0, "reward": 0.9981397390365601, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981397390365601, "reward_meter_std": 1.352704748569522e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.3527009286917746e-05, "reward_total_composite_mean": 0.9981397390365601, "reward_total_composite_std": 1.352704748569522e-05, "reward_total_mean": 0.9981397390365601, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981397390365601, "rewards/meter/std": 1.352704748569522e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981397390365601, "rewards/total_composite/std": 1.352704748569522e-05, "sampling/importance_sampling_ratio/max": 1.4492285251617432, "sampling/importance_sampling_ratio/mean": 1.0013184547424316, "sampling/importance_sampling_ratio/min": 0.6950086951255798, "sampling/sampling_logp_difference/max": 0.37103140354156494, "sampling/sampling_logp_difference/mean": 0.006174927577376366, "step": 3166 }, { "clip_ratio/high_max": 0.012976191123016179, "clip_ratio/high_mean": 0.012976191123016179, "clip_ratio/low_mean": 0.006024193484336138, "clip_ratio/low_min": 0.006024193484336138, "clip_ratio/region_mean": 0.019000384607352316, "completions/clipped_ratio": 0.0, "completions/max_length": 126.0, "completions/max_terminated_length": 126.0, "completions/mean_length": 125.125, "completions/mean_terminated_length": 125.125, "completions/min_length": 124.0, "completions/min_terminated_length": 124.0, "entropy": 0.20761017128825188, "epoch": 0.12720408081294934, "frac_reward_zero_std": 0.0, "grad_norm": 1.5397974252700806, "learning_rate": 4.0606060606060605e-07, "loss": 0.0009, "num_tokens": 7208456.0, "reward": 0.9673014283180237, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9850419759750366, "reward_meter_std": 0.024076560512185097, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.0524948425590992, "reward_total_composite_mean": 0.9673014283180237, "reward_total_composite_std": 0.052494850009679794, "reward_total_mean": 0.9673014283180237, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9850419759750366, "rewards/meter/std": 0.024076560512185097, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9673014283180237, "rewards/total_composite/std": 0.052494850009679794, "sampling/importance_sampling_ratio/max": 1.744919776916504, "sampling/importance_sampling_ratio/mean": 1.0092271566390991, "sampling/importance_sampling_ratio/min": 0.19996404647827148, "sampling/sampling_logp_difference/max": 1.6096177101135254, "sampling/sampling_logp_difference/mean": 0.02755284309387207, "step": 3167 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0007030887718428858, "epoch": 0.1272442462947343, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.030303030303031e-07, "loss": 0.0, "num_tokens": 7210280.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0013597011566162, "sampling/importance_sampling_ratio/mean": 1.000028371810913, "sampling/importance_sampling_ratio/min": 0.9954656958580017, "sampling/sampling_logp_difference/max": 0.004544587340205908, "sampling/sampling_logp_difference/mean": 8.32213117973879e-05, "step": 3168 }, { "clip_ratio/high_max": 0.028263273648917675, "clip_ratio/high_mean": 0.028263273648917675, "clip_ratio/low_mean": 0.0036231884732842445, "clip_ratio/low_min": 0.0036231884732842445, "clip_ratio/region_mean": 0.03188646212220192, "completions/clipped_ratio": 0.0, "completions/max_length": 142.0, "completions/max_terminated_length": 142.0, "completions/mean_length": 137.75, "completions/mean_terminated_length": 137.75, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.3814433552324772, "epoch": 0.12728441177651925, "frac_reward_zero_std": 0.0, "grad_norm": 2.7400386333465576, "learning_rate": 4.0000000000000003e-07, "loss": 0.0016, "num_tokens": 7212758.0, "reward": 0.973473310470581, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9911104440689087, "reward_meter_std": 0.00600889977067709, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.051604848355054855, "reward_total_composite_mean": 0.973473310470581, "reward_total_composite_std": 0.05160484462976456, "reward_total_mean": 0.973473310470581, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9911104440689087, "rewards/meter/std": 0.00600889977067709, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.973473310470581, "rewards/total_composite/std": 0.05160484462976456, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0084030628204346, "sampling/importance_sampling_ratio/min": 0.3083631992340088, "sampling/sampling_logp_difference/max": 1.1764769554138184, "sampling/sampling_logp_difference/mean": 0.04144033044576645, "step": 3169 }, { "clip_ratio/high_max": 0.0075775557197630405, "clip_ratio/high_mean": 0.0075775557197630405, "clip_ratio/low_mean": 0.004030561889521778, "clip_ratio/low_min": 0.004030561889521778, "clip_ratio/region_mean": 0.011608117609284818, "completions/clipped_ratio": 0.0, "completions/max_length": 251.0, "completions/max_terminated_length": 251.0, "completions/mean_length": 248.0, "completions/mean_terminated_length": 248.0, "completions/min_length": 244.0, "completions/min_terminated_length": 244.0, "entropy": 0.3082658052444458, "epoch": 0.1273245772583042, "frac_reward_zero_std": 0.0, "grad_norm": 1.578904628753662, "learning_rate": 3.9696969696969697e-07, "loss": 0.0069, "num_tokens": 7216446.0, "reward": 0.9604784250259399, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988981485366821, "reward_meter_std": 0.0001572021865285933, "reward_repeat_penalty_mean": 0.9615384340286255, "reward_repeat_penalty_std": 0.041117113083601, "reward_std": 0.041056081652641296, "reward_total_composite_mean": 0.9604784250259399, "reward_total_composite_std": 0.04105609655380249, "reward_total_mean": 0.9604784250259399, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988981485366821, "rewards/meter/std": 0.0001572021865285933, "rewards/repeat_penalty/mean": 0.9615384340286255, "rewards/repeat_penalty/std": 0.041117113083601, "rewards/total_composite/mean": 0.9604784250259399, "rewards/total_composite/std": 0.04105609655380249, "sampling/importance_sampling_ratio/max": 1.6617929935455322, "sampling/importance_sampling_ratio/mean": 1.0076178312301636, "sampling/importance_sampling_ratio/min": 0.248029887676239, "sampling/sampling_logp_difference/max": 1.3942060470581055, "sampling/sampling_logp_difference/mean": 0.027040995657444, "step": 3170 }, { "clip_ratio/high_max": 0.009086863487027586, "clip_ratio/high_mean": 0.009086863487027586, "clip_ratio/low_mean": 0.005335517227649689, "clip_ratio/low_min": 0.005335517227649689, "clip_ratio/region_mean": 0.014422380714677274, "completions/clipped_ratio": 0.0, "completions/max_length": 73.0, "completions/max_terminated_length": 73.0, "completions/mean_length": 69.875, "completions/mean_terminated_length": 69.875, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.1526617044582963, "epoch": 0.12736474274008916, "frac_reward_zero_std": 0.0, "grad_norm": 2.6443123817443848, "learning_rate": 3.9393939393939396e-07, "loss": -0.0007, "num_tokens": 7218213.0, "reward": 0.9944202899932861, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9944202899932861, "reward_meter_std": 0.002229472389444709, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002229455392807722, "reward_total_composite_mean": 0.9944202899932861, "reward_total_composite_std": 0.002229472389444709, "reward_total_mean": 0.9944202899932861, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9944202899932861, "rewards/meter/std": 0.002229472389444709, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9944202899932861, "rewards/total_composite/std": 0.002229472389444709, "sampling/importance_sampling_ratio/max": 1.854751706123352, "sampling/importance_sampling_ratio/mean": 1.0053551197052002, "sampling/importance_sampling_ratio/min": 0.183248370885849, "sampling/sampling_logp_difference/max": 1.6969127655029297, "sampling/sampling_logp_difference/mean": 0.02252027578651905, "step": 3171 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0011070799009758048, "epoch": 0.1274049082218741, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.9090909090909095e-07, "loss": 0.0, "num_tokens": 7219917.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0017976760864258, "sampling/importance_sampling_ratio/mean": 1.0000759363174438, "sampling/importance_sampling_ratio/min": 0.9985963106155396, "sampling/sampling_logp_difference/max": 0.0017959647811949253, "sampling/sampling_logp_difference/mean": 9.710989979794249e-05, "step": 3172 }, { "clip_ratio/high_max": 0.016817589348647743, "clip_ratio/high_mean": 0.016817589348647743, "clip_ratio/low_mean": 0.006500809220597148, "clip_ratio/low_min": 0.006500809220597148, "clip_ratio/region_mean": 0.02331839856924489, "completions/clipped_ratio": 0.0, "completions/max_length": 234.0, "completions/max_terminated_length": 234.0, "completions/mean_length": 230.75, "completions/mean_terminated_length": 230.75, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.2557064164429903, "epoch": 0.12744507370365907, "frac_reward_zero_std": 0.0, "grad_norm": 1.6205352544784546, "learning_rate": 3.878787878787879e-07, "loss": 0.0057, "num_tokens": 7223459.0, "reward": 0.9605352282524109, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989705085754395, "reward_meter_std": 0.00055393495131284, "reward_repeat_penalty_mean": 0.9615384340286255, "reward_repeat_penalty_std": 0.058148376643657684, "reward_std": 0.05785060301423073, "reward_total_composite_mean": 0.9605352282524109, "reward_total_composite_std": 0.05785057693719864, "reward_total_mean": 0.9605352282524109, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989705085754395, "rewards/meter/std": 0.00055393495131284, "rewards/repeat_penalty/mean": 0.9615384340286255, "rewards/repeat_penalty/std": 0.058148376643657684, "rewards/total_composite/mean": 0.9605352282524109, "rewards/total_composite/std": 0.05785057693719864, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0043230056762695, "sampling/importance_sampling_ratio/min": 0.3292386531829834, "sampling/sampling_logp_difference/max": 1.1109724044799805, "sampling/sampling_logp_difference/mean": 0.025763755664229393, "step": 3173 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 36.0, "completions/mean_terminated_length": 36.0, "completions/min_length": 36.0, "completions/min_terminated_length": 36.0, "entropy": 0.016203245264478028, "epoch": 0.12748523918544402, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.848484848484849e-07, "loss": 0.0, "num_tokens": 7224971.0, "reward": 0.9996045231819153, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9996045231819153, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9996045231819153, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9996045231819153, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9996045231819153, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9996045231819153, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0477015972137451, "sampling/importance_sampling_ratio/mean": 1.0015071630477905, "sampling/importance_sampling_ratio/min": 0.9647499918937683, "sampling/sampling_logp_difference/max": 0.04659882187843323, "sampling/sampling_logp_difference/mean": 0.0017466507852077484, "step": 3174 }, { "clip_ratio/high_max": 0.0018939394503831863, "clip_ratio/high_mean": 0.0018939394503831863, "clip_ratio/low_mean": 0.0028337890980765224, "clip_ratio/low_min": 0.0028337890980765224, "clip_ratio/region_mean": 0.004727728548459709, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 132.125, "completions/mean_terminated_length": 132.125, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.08139224071055651, "epoch": 0.12752540466722898, "frac_reward_zero_std": 0.0, "grad_norm": 0.08287062495946884, "learning_rate": 3.8181818181818187e-07, "loss": -0.0001, "num_tokens": 7227420.0, "reward": 0.9994284510612488, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994284510612488, "reward_meter_std": 6.205716090335045e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.210763785929885e-06, "reward_total_composite_mean": 0.9994284510612488, "reward_total_composite_std": 6.205716090335045e-06, "reward_total_mean": 0.9994284510612488, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994284510612488, "rewards/meter/std": 6.205716090335045e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994284510612488, "rewards/total_composite/std": 6.205716090335045e-06, "sampling/importance_sampling_ratio/max": 1.300461769104004, "sampling/importance_sampling_ratio/mean": 1.0014747381210327, "sampling/importance_sampling_ratio/min": 0.3827598989009857, "sampling/sampling_logp_difference/max": 0.9603474140167236, "sampling/sampling_logp_difference/mean": 0.008646445348858833, "step": 3175 }, { "clip_ratio/high_max": 0.012750733643770218, "clip_ratio/high_mean": 0.012750733643770218, "clip_ratio/low_mean": 0.008392618678044528, "clip_ratio/low_min": 0.008392618678044528, "clip_ratio/region_mean": 0.021143352321814746, "completions/clipped_ratio": 0.0, "completions/max_length": 371.0, "completions/max_terminated_length": 371.0, "completions/mean_length": 351.0, "completions/mean_terminated_length": 351.0, "completions/min_length": 317.0, "completions/min_terminated_length": 317.0, "entropy": 0.3763493411242962, "epoch": 0.12756557014901393, "frac_reward_zero_std": 0.0, "grad_norm": 1.365169644355774, "learning_rate": 3.787878787878788e-07, "loss": -0.0238, "num_tokens": 7232148.0, "reward": 0.9475728273391724, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9750000238418579, "reward_count_adherence_std": 0.04629101976752281, "reward_meter_mean": 0.9988387823104858, "reward_meter_std": 0.0003177140897605568, "reward_repeat_penalty_mean": 0.9736841917037964, "reward_repeat_penalty_std": 0.03978573530912399, "reward_std": 0.04686008021235466, "reward_total_composite_mean": 0.9475728273391724, "reward_total_composite_std": 0.04686007648706436, "reward_total_mean": 0.9475728273391724, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9750000238418579, "rewards/count_adherence/std": 0.04629101976752281, "rewards/meter/mean": 0.9988387823104858, "rewards/meter/std": 0.0003177140897605568, "rewards/repeat_penalty/mean": 0.9736841917037964, "rewards/repeat_penalty/std": 0.03978573530912399, "rewards/total_composite/mean": 0.9475728273391724, "rewards/total_composite/std": 0.04686007648706436, "sampling/importance_sampling_ratio/max": 1.7764185667037964, "sampling/importance_sampling_ratio/mean": 1.008555293083191, "sampling/importance_sampling_ratio/min": 0.007433735765516758, "sampling/sampling_logp_difference/max": 4.901726722717285, "sampling/sampling_logp_difference/mean": 0.03374847397208214, "step": 3176 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.001803019520593807, "epoch": 0.12760573563079888, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.757575757575758e-07, "loss": 0.0, "num_tokens": 7233836.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.002587914466858, "sampling/importance_sampling_ratio/mean": 1.0002270936965942, "sampling/importance_sampling_ratio/min": 0.9966246485710144, "sampling/sampling_logp_difference/max": 0.0033810948953032494, "sampling/sampling_logp_difference/mean": 0.00024669314734637737, "step": 3177 }, { "clip_ratio/high_max": 0.0034966744715347886, "clip_ratio/high_mean": 0.0034966744715347886, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0034966744715347886, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.25, "completions/mean_terminated_length": 71.25, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.07951132394373417, "epoch": 0.12764590111258384, "frac_reward_zero_std": 0.0, "grad_norm": 1.7416008710861206, "learning_rate": 3.7272727272727274e-07, "loss": 0.0007, "num_tokens": 7235742.0, "reward": 0.9994281530380249, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994281530380249, "reward_meter_std": 4.47747042926494e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 4.4774700654670596e-05, "reward_total_composite_mean": 0.9994281530380249, "reward_total_composite_std": 4.47747042926494e-05, "reward_total_mean": 0.9994281530380249, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994281530380249, "rewards/meter/std": 4.47747042926494e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994281530380249, "rewards/total_composite/std": 4.47747042926494e-05, "sampling/importance_sampling_ratio/max": 1.2836189270019531, "sampling/importance_sampling_ratio/mean": 1.004219651222229, "sampling/importance_sampling_ratio/min": 0.691192090511322, "sampling/sampling_logp_difference/max": 0.3693375587463379, "sampling/sampling_logp_difference/mean": 0.007580549921840429, "step": 3178 }, { "clip_ratio/high_max": 0.03241610305849463, "clip_ratio/high_mean": 0.03241610305849463, "clip_ratio/low_mean": 0.00682031805627048, "clip_ratio/low_min": 0.00682031805627048, "clip_ratio/region_mean": 0.03923642111476511, "completions/clipped_ratio": 0.0, "completions/max_length": 241.0, "completions/max_terminated_length": 241.0, "completions/mean_length": 234.75, "completions/mean_terminated_length": 234.75, "completions/min_length": 224.0, "completions/min_terminated_length": 224.0, "entropy": 0.37664885073900223, "epoch": 0.1276860665943688, "frac_reward_zero_std": 0.0, "grad_norm": 2.992551565170288, "learning_rate": 3.6969696969696973e-07, "loss": 0.0176, "num_tokens": 7239068.0, "reward": 0.9794449806213379, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998660683631897, "reward_meter_std": 0.0007363483309745789, "reward_repeat_penalty_mean": 0.9807692170143127, "reward_repeat_penalty_std": 0.03560846298933029, "reward_std": 0.03524333983659744, "reward_total_composite_mean": 0.9794449806213379, "reward_total_composite_std": 0.03524334356188774, "reward_total_mean": 0.9794449806213379, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998660683631897, "rewards/meter/std": 0.0007363483309745789, "rewards/repeat_penalty/mean": 0.9807692170143127, "rewards/repeat_penalty/std": 0.03560846298933029, "rewards/total_composite/mean": 0.9794449806213379, "rewards/total_composite/std": 0.03524334356188774, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0093677043914795, "sampling/importance_sampling_ratio/min": 0.03618064150214195, "sampling/sampling_logp_difference/max": 3.3192310333251953, "sampling/sampling_logp_difference/mean": 0.050825342535972595, "step": 3179 }, { "clip_ratio/high_max": 0.002848137926775962, "clip_ratio/high_mean": 0.002848137926775962, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.002848137926775962, "completions/clipped_ratio": 0.0, "completions/max_length": 133.0, "completions/max_terminated_length": 133.0, "completions/mean_length": 131.875, "completions/mean_terminated_length": 131.875, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.08243910130113363, "epoch": 0.12772623207615375, "frac_reward_zero_std": 0.0, "grad_norm": 0.22532886266708374, "learning_rate": 3.666666666666667e-07, "loss": -0.0004, "num_tokens": 7241707.0, "reward": 0.9994198083877563, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994198083877563, "reward_meter_std": 2.31738686125027e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.3179616619017906e-05, "reward_total_composite_mean": 0.9994198083877563, "reward_total_composite_std": 2.31738686125027e-05, "reward_total_mean": 0.9994198083877563, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994198083877563, "rewards/meter/std": 2.31738686125027e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994198083877563, "rewards/total_composite/std": 2.31738686125027e-05, "sampling/importance_sampling_ratio/max": 1.3655613660812378, "sampling/importance_sampling_ratio/mean": 1.0032905340194702, "sampling/importance_sampling_ratio/min": 0.2511071562767029, "sampling/sampling_logp_difference/max": 1.3818755149841309, "sampling/sampling_logp_difference/mean": 0.010175392031669617, "step": 3180 }, { "clip_ratio/high_max": 0.004531931248493493, "clip_ratio/high_mean": 0.004531931248493493, "clip_ratio/low_mean": 0.0030676000751554966, "clip_ratio/low_min": 0.0030676000751554966, "clip_ratio/region_mean": 0.007599531323648989, "completions/clipped_ratio": 0.0, "completions/max_length": 167.0, "completions/max_terminated_length": 167.0, "completions/mean_length": 165.0, "completions/mean_terminated_length": 165.0, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.12881212681531906, "epoch": 0.1277663975579387, "frac_reward_zero_std": 0.0, "grad_norm": 0.8209965825080872, "learning_rate": 3.6363636363636366e-07, "loss": -0.0034, "num_tokens": 7244435.0, "reward": 0.9993767738342285, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993767738342285, "reward_meter_std": 0.00010731098882388324, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00010731603106250986, "reward_total_composite_mean": 0.9993767738342285, "reward_total_composite_std": 0.00010731098882388324, "reward_total_mean": 0.9993767738342285, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993767738342285, "rewards/meter/std": 0.00010731098882388324, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993767738342285, "rewards/total_composite/std": 0.00010731098882388324, "sampling/importance_sampling_ratio/max": 1.6672699451446533, "sampling/importance_sampling_ratio/mean": 1.0039643049240112, "sampling/importance_sampling_ratio/min": 0.31161612272262573, "sampling/sampling_logp_difference/max": 1.1659832000732422, "sampling/sampling_logp_difference/mean": 0.014372244477272034, "step": 3181 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.005434782709926367, "clip_ratio/low_min": 0.005434782709926367, "clip_ratio/region_mean": 0.005434782709926367, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.05426653893664479, "epoch": 0.12780656303972365, "frac_reward_zero_std": 0.0, "grad_norm": 1.5534006357192993, "learning_rate": 3.6060606060606065e-07, "loss": -0.0002, "num_tokens": 7246300.0, "reward": 0.9994250535964966, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994250535964966, "reward_meter_std": 0.00022647925652563572, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002264756039949134, "reward_total_composite_mean": 0.9994250535964966, "reward_total_composite_std": 0.00022647925652563572, "reward_total_mean": 0.9994250535964966, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994250535964966, "rewards/meter/std": 0.00022647925652563572, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994250535964966, "rewards/total_composite/std": 0.00022647925652563572, "sampling/importance_sampling_ratio/max": 1.234535813331604, "sampling/importance_sampling_ratio/mean": 1.0001282691955566, "sampling/importance_sampling_ratio/min": 0.5230568647384644, "sampling/sampling_logp_difference/max": 0.6480650901794434, "sampling/sampling_logp_difference/mean": 0.007597931195050478, "step": 3182 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 35.0, "completions/max_terminated_length": 35.0, "completions/mean_length": 35.0, "completions/mean_terminated_length": 35.0, "completions/min_length": 35.0, "completions/min_terminated_length": 35.0, "entropy": 0.0031260672258213162, "epoch": 0.1278467285215086, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.5757575757575764e-07, "loss": 0.0, "num_tokens": 7247884.0, "reward": 0.9980231523513794, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980231523513794, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9980231523513794, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9980231523513794, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980231523513794, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9980231523513794, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0085147619247437, "sampling/importance_sampling_ratio/mean": 1.0004373788833618, "sampling/importance_sampling_ratio/min": 1.000000238418579, "sampling/sampling_logp_difference/max": 0.008478658273816109, "sampling/sampling_logp_difference/mean": 0.00043650183943100274, "step": 3183 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.0018656715983524919, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.04700829554349184, "epoch": 0.12788689400329356, "frac_reward_zero_std": 0.0, "grad_norm": 0.10775468498468399, "learning_rate": 3.545454545454546e-07, "loss": -0.0004, "num_tokens": 7249747.0, "reward": 0.9981533288955688, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981533288955688, "reward_meter_std": 3.6652184007834876e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.6725155041494872e-06, "reward_total_composite_mean": 0.9981533288955688, "reward_total_composite_std": 3.6652184007834876e-06, "reward_total_mean": 0.9981533288955688, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981533288955688, "rewards/meter/std": 3.6652184007834876e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981533288955688, "rewards/total_composite/std": 3.6652184007834876e-06, "sampling/importance_sampling_ratio/max": 1.437882661819458, "sampling/importance_sampling_ratio/mean": 1.000977635383606, "sampling/importance_sampling_ratio/min": 0.4049831032752991, "sampling/sampling_logp_difference/max": 0.9039099216461182, "sampling/sampling_logp_difference/mean": 0.007368480786681175, "step": 3184 }, { "clip_ratio/high_max": 0.013145374483428895, "clip_ratio/high_mean": 0.013145374483428895, "clip_ratio/low_mean": 0.0071428571827709675, "clip_ratio/low_min": 0.0071428571827709675, "clip_ratio/region_mean": 0.020288231666199863, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 66.25, "completions/mean_terminated_length": 66.25, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.15871062874794006, "epoch": 0.12792705948507851, "frac_reward_zero_std": 0.0, "grad_norm": 5.8911333084106445, "learning_rate": 3.515151515151515e-07, "loss": 0.0221, "num_tokens": 7251597.0, "reward": 0.9895763397216797, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9895763397216797, "reward_meter_std": 0.009416628628969193, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00941664818674326, "reward_total_composite_mean": 0.9895763397216797, "reward_total_composite_std": 0.009416628628969193, "reward_total_mean": 0.9895763397216797, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9895763397216797, "rewards/meter/std": 0.009416628628969193, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9895763397216797, "rewards/total_composite/std": 0.009416628628969193, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0064246654510498, "sampling/importance_sampling_ratio/min": 0.1616344153881073, "sampling/sampling_logp_difference/max": 1.822418212890625, "sampling/sampling_logp_difference/mean": 0.02673407271504402, "step": 3185 }, { "clip_ratio/high_max": 0.022629241226240993, "clip_ratio/high_mean": 0.022629241226240993, "clip_ratio/low_mean": 0.005434782709926367, "clip_ratio/low_min": 0.005434782709926367, "clip_ratio/region_mean": 0.02806402393616736, "completions/clipped_ratio": 0.0, "completions/max_length": 47.0, "completions/max_terminated_length": 47.0, "completions/mean_length": 45.75, "completions/mean_terminated_length": 45.75, "completions/min_length": 43.0, "completions/min_terminated_length": 43.0, "entropy": 0.14682762045413256, "epoch": 0.12796722496686347, "frac_reward_zero_std": 0.0, "grad_norm": 7.721897125244141, "learning_rate": 3.4848484848484856e-07, "loss": 0.0207, "num_tokens": 7253219.0, "reward": 0.9423834681510925, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9423834681510925, "reward_meter_std": 0.005602226592600346, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.005602239165455103, "reward_total_composite_mean": 0.9423834681510925, "reward_total_composite_std": 0.005602226592600346, "reward_total_mean": 0.9423834681510925, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9423834681510925, "rewards/meter/std": 0.005602226592600346, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9423834681510925, "rewards/total_composite/std": 0.005602226592600346, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0041768550872803, "sampling/importance_sampling_ratio/min": 0.1787530481815338, "sampling/sampling_logp_difference/max": 1.721750020980835, "sampling/sampling_logp_difference/mean": 0.047884196043014526, "step": 3186 }, { "clip_ratio/high_max": 0.00957481365185231, "clip_ratio/high_mean": 0.00957481365185231, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.00957481365185231, "completions/clipped_ratio": 0.0, "completions/max_length": 81.0, "completions/max_terminated_length": 81.0, "completions/mean_length": 78.625, "completions/mean_terminated_length": 78.625, "completions/min_length": 78.0, "completions/min_terminated_length": 78.0, "entropy": 0.17396301217377186, "epoch": 0.12800739044864842, "frac_reward_zero_std": 0.0, "grad_norm": 1.325953483581543, "learning_rate": 3.454545454545455e-07, "loss": 0.0047, "num_tokens": 7255248.0, "reward": 0.9991061091423035, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991061091423035, "reward_meter_std": 8.766540122451261e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.767841063672677e-05, "reward_total_composite_mean": 0.9991061091423035, "reward_total_composite_std": 8.766540122451261e-05, "reward_total_mean": 0.9991061091423035, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991061091423035, "rewards/meter/std": 8.766540122451261e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991061091423035, "rewards/total_composite/std": 8.766540122451261e-05, "sampling/importance_sampling_ratio/max": 1.395779013633728, "sampling/importance_sampling_ratio/mean": 1.0035501718521118, "sampling/importance_sampling_ratio/min": 0.3711963891983032, "sampling/sampling_logp_difference/max": 0.9910240173339844, "sampling/sampling_logp_difference/mean": 0.01794569194316864, "step": 3187 }, { "clip_ratio/high_max": 0.014866774436086416, "clip_ratio/high_mean": 0.014866774436086416, "clip_ratio/low_mean": 0.0012499999720603228, "clip_ratio/low_min": 0.0012499999720603228, "clip_ratio/region_mean": 0.01611677440814674, "completions/clipped_ratio": 0.0, "completions/max_length": 103.0, "completions/max_terminated_length": 103.0, "completions/mean_length": 100.75, "completions/mean_terminated_length": 100.75, "completions/min_length": 99.0, "completions/min_terminated_length": 99.0, "entropy": 0.1316568348556757, "epoch": 0.12804755593043338, "frac_reward_zero_std": 0.0, "grad_norm": 2.738718032836914, "learning_rate": 3.4242424242424243e-07, "loss": 0.0022, "num_tokens": 7257350.0, "reward": 0.9742865562438965, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992691278457642, "reward_meter_std": 0.00017112072964664549, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07064837962388992, "reward_total_composite_mean": 0.9742865562438965, "reward_total_composite_std": 0.07064837217330933, "reward_total_mean": 0.9742865562438965, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992691278457642, "rewards/meter/std": 0.00017112072964664549, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9742865562438965, "rewards/total_composite/std": 0.07064837217330933, "sampling/importance_sampling_ratio/max": 1.7092481851577759, "sampling/importance_sampling_ratio/mean": 1.0054959058761597, "sampling/importance_sampling_ratio/min": 0.20339646935462952, "sampling/sampling_logp_difference/max": 1.5925981998443604, "sampling/sampling_logp_difference/mean": 0.018476763740181923, "step": 3188 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 34.0, "completions/mean_terminated_length": 34.0, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.03649881505407393, "epoch": 0.12808772141221833, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.393939393939395e-07, "loss": 0.0, "num_tokens": 7258606.0, "reward": 0.9923644065856934, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9923644065856934, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9923644065856934, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9923644065856934, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9923644065856934, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9923644065856934, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.097429871559143, "sampling/importance_sampling_ratio/mean": 1.0026298761367798, "sampling/importance_sampling_ratio/min": 0.961505115032196, "sampling/sampling_logp_difference/max": 0.09297092258930206, "sampling/sampling_logp_difference/mean": 0.003486229106783867, "step": 3189 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.0018382353009656072, "completions/clipped_ratio": 0.0, "completions/max_length": 70.0, "completions/max_terminated_length": 70.0, "completions/mean_length": 68.25, "completions/mean_terminated_length": 68.25, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.052496086340397596, "epoch": 0.12812788689400328, "frac_reward_zero_std": 0.0, "grad_norm": 2.772718906402588, "learning_rate": 3.363636363636364e-07, "loss": 0.0015, "num_tokens": 7260528.0, "reward": 0.9994656443595886, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994656443595886, "reward_meter_std": 8.461968536721542e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.462517871521413e-05, "reward_total_composite_mean": 0.9994656443595886, "reward_total_composite_std": 8.461968536721542e-05, "reward_total_mean": 0.9994656443595886, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994656443595886, "rewards/meter/std": 8.461968536721542e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994656443595886, "rewards/total_composite/std": 8.461968536721542e-05, "sampling/importance_sampling_ratio/max": 1.5606142282485962, "sampling/importance_sampling_ratio/mean": 1.003930926322937, "sampling/importance_sampling_ratio/min": 0.39883288741111755, "sampling/sampling_logp_difference/max": 0.919212818145752, "sampling/sampling_logp_difference/mean": 0.00850264448672533, "step": 3190 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 71.0, "completions/mean_terminated_length": 71.0, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.08830153662711382, "epoch": 0.12816805237578824, "frac_reward_zero_std": 0.0, "grad_norm": 0.23446327447891235, "learning_rate": 3.3333333333333335e-07, "loss": 0.0002, "num_tokens": 7262336.0, "reward": 0.9994317293167114, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994317293167114, "reward_meter_std": 1.128076564782532e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.127942687162431e-05, "reward_total_composite_mean": 0.9994317293167114, "reward_total_composite_std": 1.128076564782532e-05, "reward_total_mean": 0.9994317293167114, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994317293167114, "rewards/meter/std": 1.128076564782532e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994317293167114, "rewards/total_composite/std": 1.128076564782532e-05, "sampling/importance_sampling_ratio/max": 1.223402500152588, "sampling/importance_sampling_ratio/mean": 1.0047414302825928, "sampling/importance_sampling_ratio/min": 0.7457441091537476, "sampling/sampling_logp_difference/max": 0.2933727502822876, "sampling/sampling_logp_difference/mean": 0.007548379711806774, "step": 3191 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0069444444961845875, "clip_ratio/low_min": 0.0069444444961845875, "clip_ratio/region_mean": 0.0069444444961845875, "completions/clipped_ratio": 0.0, "completions/max_length": 36.0, "completions/max_terminated_length": 36.0, "completions/mean_length": 34.25, "completions/mean_terminated_length": 34.25, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.05123408907093108, "epoch": 0.1282082178575732, "frac_reward_zero_std": 0.0, "grad_norm": 5.140721321105957, "learning_rate": 3.303030303030303e-07, "loss": 0.0136, "num_tokens": 7263826.0, "reward": 0.9904930591583252, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9904930591583252, "reward_meter_std": 0.02051379904150963, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.02051379904150963, "reward_total_composite_mean": 0.9904930591583252, "reward_total_composite_std": 0.02051379904150963, "reward_total_mean": 0.9904930591583252, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9904930591583252, "rewards/meter/std": 0.02051379904150963, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9904930591583252, "rewards/total_composite/std": 0.02051379904150963, "sampling/importance_sampling_ratio/max": 1.33383309841156, "sampling/importance_sampling_ratio/mean": 1.0013935565948486, "sampling/importance_sampling_ratio/min": 0.4425126910209656, "sampling/sampling_logp_difference/max": 0.8152861595153809, "sampling/sampling_logp_difference/mean": 0.006564823444932699, "step": 3192 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0020491802133619785, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.013490123208612204, "epoch": 0.12824838333935815, "frac_reward_zero_std": 0.0, "grad_norm": 0.0006765525904484093, "learning_rate": 3.2727272727272733e-07, "loss": 0.0001, "num_tokens": 7265538.0, "reward": 0.9973390102386475, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973390102386475, "reward_meter_std": 2.954577098535083e-07, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.0140995477268007e-07, "reward_total_composite_mean": 0.9973390102386475, "reward_total_composite_std": 2.954577098535083e-07, "reward_total_mean": 0.9973390102386475, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973390102386475, "rewards/meter/std": 2.954577098535083e-07, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973390102386475, "rewards/total_composite/std": 2.954577098535083e-07, "sampling/importance_sampling_ratio/max": 1.0959326028823853, "sampling/importance_sampling_ratio/mean": 1.0007272958755493, "sampling/importance_sampling_ratio/min": 0.9242770671844482, "sampling/sampling_logp_difference/max": 0.09160566329956055, "sampling/sampling_logp_difference/mean": 0.0017861899686977267, "step": 3193 }, { "clip_ratio/high_max": 0.006628788076341152, "clip_ratio/high_mean": 0.006628788076341152, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.006628788076341152, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 131.875, "completions/mean_terminated_length": 131.875, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.0831528240814805, "epoch": 0.1282885488211431, "frac_reward_zero_std": 0.0, "grad_norm": 0.48764580488204956, "learning_rate": 3.2424242424242427e-07, "loss": -0.0017, "num_tokens": 7267985.0, "reward": 0.999407172203064, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999407172203064, "reward_meter_std": 5.462394256028347e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.462476474349387e-05, "reward_total_composite_mean": 0.999407172203064, "reward_total_composite_std": 5.462394256028347e-05, "reward_total_mean": 0.999407172203064, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999407172203064, "rewards/meter/std": 5.462394256028347e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999407172203064, "rewards/total_composite/std": 5.462394256028347e-05, "sampling/importance_sampling_ratio/max": 1.5405017137527466, "sampling/importance_sampling_ratio/mean": 1.003739833831787, "sampling/importance_sampling_ratio/min": 0.5363329648971558, "sampling/sampling_logp_difference/max": 0.6230001449584961, "sampling/sampling_logp_difference/mean": 0.0077711548656225204, "step": 3194 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.000906125656911172, "epoch": 0.12832871430292805, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.212121212121212e-07, "loss": 0.0, "num_tokens": 7269473.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.001376748085022, "sampling/importance_sampling_ratio/mean": 1.0000648498535156, "sampling/importance_sampling_ratio/min": 0.9994438290596008, "sampling/sampling_logp_difference/max": 0.0013757887063547969, "sampling/sampling_logp_difference/mean": 6.992343696765602e-05, "step": 3195 }, { "clip_ratio/high_max": 0.007785308640450239, "clip_ratio/high_mean": 0.007785308640450239, "clip_ratio/low_mean": 0.017221659421920776, "clip_ratio/low_min": 0.017221659421920776, "clip_ratio/region_mean": 0.025006968062371016, "completions/clipped_ratio": 0.0, "completions/max_length": 424.0, "completions/max_terminated_length": 424.0, "completions/mean_length": 397.5, "completions/mean_terminated_length": 397.5, "completions/min_length": 379.0, "completions/min_terminated_length": 379.0, "entropy": 0.5010012574493885, "epoch": 0.128368879784713, "frac_reward_zero_std": 0.0, "grad_norm": 1.7959932088851929, "learning_rate": 3.181818181818182e-07, "loss": -0.0218, "num_tokens": 7274421.0, "reward": 0.782776951789856, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7884615659713745, "reward_count_adherence_std": 0.03560846298933029, "reward_meter_mean": 0.9988911747932434, "reward_meter_std": 0.0005708896787837148, "reward_repeat_penalty_mean": 0.9937499761581421, "reward_repeat_penalty_std": 0.01767767407000065, "reward_std": 0.04057204723358154, "reward_total_composite_mean": 0.782776951789856, "reward_total_composite_std": 0.04057204723358154, "reward_total_mean": 0.782776951789856, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7884615659713745, "rewards/count_adherence/std": 0.03560846298933029, "rewards/meter/mean": 0.9988911747932434, "rewards/meter/std": 0.0005708896787837148, "rewards/repeat_penalty/mean": 0.9937499761581421, "rewards/repeat_penalty/std": 0.01767767407000065, "rewards/total_composite/mean": 0.782776951789856, "rewards/total_composite/std": 0.04057204723358154, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0130902528762817, "sampling/importance_sampling_ratio/min": 0.15019360184669495, "sampling/sampling_logp_difference/max": 1.8958301544189453, "sampling/sampling_logp_difference/mean": 0.05620747059583664, "step": 3196 }, { "clip_ratio/high_max": 0.036240562330931425, "clip_ratio/high_mean": 0.036240562330931425, "clip_ratio/low_mean": 0.012125813402235508, "clip_ratio/low_min": 0.012125813402235508, "clip_ratio/region_mean": 0.04836637573316693, "completions/clipped_ratio": 0.0, "completions/max_length": 313.0, "completions/max_terminated_length": 313.0, "completions/mean_length": 277.375, "completions/mean_terminated_length": 277.375, "completions/min_length": 254.0, "completions/min_terminated_length": 254.0, "entropy": 0.570828665047884, "epoch": 0.12840904526649796, "frac_reward_zero_std": 0.0, "grad_norm": 3.9520342350006104, "learning_rate": 3.151515151515152e-07, "loss": 0.0732, "num_tokens": 7278424.0, "reward": 0.953273355960846, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.96875, "reward_count_adherence_std": 0.0578637570142746, "reward_meter_mean": 0.9972818493843079, "reward_meter_std": 0.001952509512193501, "reward_repeat_penalty_mean": 0.9852941036224365, "reward_repeat_penalty_std": 0.04159451276063919, "reward_std": 0.08584806323051453, "reward_total_composite_mean": 0.953273355960846, "reward_total_composite_std": 0.08584806323051453, "reward_total_mean": 0.953273355960846, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.96875, "rewards/count_adherence/std": 0.0578637570142746, "rewards/meter/mean": 0.9972818493843079, "rewards/meter/std": 0.001952509512193501, "rewards/repeat_penalty/mean": 0.9852941036224365, "rewards/repeat_penalty/std": 0.04159451276063919, "rewards/total_composite/mean": 0.953273355960846, "rewards/total_composite/std": 0.08584806323051453, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01344633102417, "sampling/importance_sampling_ratio/min": 0.22898982465267181, "sampling/sampling_logp_difference/max": 1.7296597957611084, "sampling/sampling_logp_difference/mean": 0.06301229447126389, "step": 3197 }, { "clip_ratio/high_max": 0.019021739484742284, "clip_ratio/high_mean": 0.019021739484742284, "clip_ratio/low_mean": 0.013334879651665688, "clip_ratio/low_min": 0.013334879651665688, "clip_ratio/region_mean": 0.03235661913640797, "completions/clipped_ratio": 0.0, "completions/max_length": 49.0, "completions/max_terminated_length": 49.0, "completions/mean_length": 46.125, "completions/mean_terminated_length": 46.125, "completions/min_length": 44.0, "completions/min_terminated_length": 44.0, "entropy": 0.14615605771541595, "epoch": 0.12844921074828292, "frac_reward_zero_std": 0.0, "grad_norm": 11.894042015075684, "learning_rate": 3.1212121212121213e-07, "loss": 0.0029, "num_tokens": 7280073.0, "reward": 0.9368916749954224, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9368916749954224, "reward_meter_std": 0.018891815096139908, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.018891818821430206, "reward_total_composite_mean": 0.9368916749954224, "reward_total_composite_std": 0.018891815096139908, "reward_total_mean": 0.9368916749954224, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9368916749954224, "rewards/meter/std": 0.018891815096139908, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9368916749954224, "rewards/total_composite/std": 0.018891815096139908, "sampling/importance_sampling_ratio/max": 1.6582306623458862, "sampling/importance_sampling_ratio/mean": 1.0053000450134277, "sampling/importance_sampling_ratio/min": 0.10954440385103226, "sampling/sampling_logp_difference/max": 2.211425304412842, "sampling/sampling_logp_difference/mean": 0.03293498232960701, "step": 3198 }, { "clip_ratio/high_max": 0.010650901531334966, "clip_ratio/high_mean": 0.010650901531334966, "clip_ratio/low_mean": 0.00802644295617938, "clip_ratio/low_min": 0.00802644295617938, "clip_ratio/region_mean": 0.018677344487514347, "completions/clipped_ratio": 0.0, "completions/max_length": 207.0, "completions/max_terminated_length": 207.0, "completions/mean_length": 201.875, "completions/mean_terminated_length": 201.875, "completions/min_length": 189.0, "completions/min_terminated_length": 189.0, "entropy": 0.31216721422970295, "epoch": 0.12848937623006787, "frac_reward_zero_std": 0.0, "grad_norm": 1.7074098587036133, "learning_rate": 3.090909090909091e-07, "loss": 0.0104, "num_tokens": 7283464.0, "reward": 0.7810623049736023, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.8017416000366211, "reward_meter_std": 0.2793266177177429, "reward_repeat_penalty_mean": 0.96875, "reward_repeat_penalty_std": 0.0431290864944458, "reward_std": 0.2836529314517975, "reward_total_composite_mean": 0.7810623049736023, "reward_total_composite_std": 0.2836529314517975, "reward_total_mean": 0.7810623049736023, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.8017416000366211, "rewards/meter/std": 0.2793266177177429, "rewards/repeat_penalty/mean": 0.96875, "rewards/repeat_penalty/std": 0.0431290864944458, "rewards/total_composite/mean": 0.7810623049736023, "rewards/total_composite/std": 0.2836529314517975, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.007271647453308, "sampling/importance_sampling_ratio/min": 0.009904325008392334, "sampling/sampling_logp_difference/max": 4.614783763885498, "sampling/sampling_logp_difference/mean": 0.032916199415922165, "step": 3199 }, { "clip_ratio/high_max": 0.004032258060760796, "clip_ratio/high_mean": 0.004032258060760796, "clip_ratio/low_mean": 0.0026881720405071974, "clip_ratio/low_min": 0.0026881720405071974, "clip_ratio/region_mean": 0.0067204301012679935, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 93.0, "completions/mean_terminated_length": 93.0, "completions/min_length": 93.0, "completions/min_terminated_length": 93.0, "entropy": 0.07502706721425056, "epoch": 0.12852954171185282, "frac_reward_zero_std": 0.0, "grad_norm": 0.4144695997238159, "learning_rate": 3.0606060606060606e-07, "loss": 0.0, "num_tokens": 7285520.0, "reward": 0.9977416396141052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977416396141052, "reward_meter_std": 3.231431037420407e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.2314139389200136e-05, "reward_total_composite_mean": 0.9977416396141052, "reward_total_composite_std": 3.231431037420407e-05, "reward_total_mean": 0.9977416396141052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977416396141052, "rewards/meter/std": 3.231431037420407e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977416396141052, "rewards/total_composite/std": 3.231431037420407e-05, "sampling/importance_sampling_ratio/max": 1.8610354661941528, "sampling/importance_sampling_ratio/mean": 1.0042681694030762, "sampling/importance_sampling_ratio/min": 0.5517995953559875, "sampling/sampling_logp_difference/max": 0.6211330890655518, "sampling/sampling_logp_difference/mean": 0.010148909874260426, "step": 3200 }, { "epoch": 0.12852954171185282, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.04807692307692308, "eval_completions/max_length": 424.15384615384613, "eval_completions/max_terminated_length": 381.38461538461536, "eval_completions/mean_length": 217.47115384615384, "eval_completions/mean_terminated_length": 202.3791222205529, "eval_completions/min_length": 60.76923076923077, "eval_completions/min_terminated_length": 60.76923076923077, "eval_entropy": 0.38261800087415254, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 7285520.0, "eval_reward": 0.716111809015274, "eval_reward_arabic_clean_mean": 0.9711538461538461, "eval_reward_arabic_clean_std": 0.06280488005051246, "eval_reward_count_adherence_mean": 0.962191457931812, "eval_reward_count_adherence_std": 0.056447364103335604, "eval_reward_meter_mean": 0.7966891756424537, "eval_reward_meter_std": 0.3122874472576838, "eval_reward_repeat_penalty_mean": 0.9550704268308786, "eval_reward_repeat_penalty_std": 0.07092011161148548, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.716111809015274, "eval_reward_total_composite_std": 0.33144921350937623, "eval_reward_total_mean": 0.716111809015274, "eval_rewards/arabic_clean/mean": 0.9711538461538461, "eval_rewards/arabic_clean/std": 0.06280488005051246, "eval_rewards/count_adherence/mean": 0.962191457931812, "eval_rewards/count_adherence/std": 0.056447364103335604, "eval_rewards/meter/mean": 0.7966891756424537, "eval_rewards/meter/std": 0.3122874472576838, "eval_rewards/repeat_penalty/mean": 0.9550704268308786, "eval_rewards/repeat_penalty/std": 0.07092011161148548, "eval_rewards/total_composite/mean": 0.716111809015274, "eval_rewards/total_composite/std": 0.33144921350937623, "eval_runtime": 78.3425, "eval_samples_per_second": 1.328, "eval_sampling/importance_sampling_ratio/max": 1.5303410291671753, "eval_sampling/importance_sampling_ratio/mean": 1.0096190892733061, "eval_sampling/importance_sampling_ratio/min": 0.3024700696651752, "eval_sampling/sampling_logp_difference/max": 1.224547532888559, "eval_sampling/sampling_logp_difference/mean": 0.03270428670713535, "eval_steps_per_second": 0.166, "step": 3200 }, { "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/low_mean": 0.005281690042465925, "clip_ratio/low_min": 0.005281690042465925, "clip_ratio/region_mean": 0.007042253389954567, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 71.0, "completions/mean_terminated_length": 71.0, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.07758000679314137, "epoch": 0.12856970719363778, "frac_reward_zero_std": 0.0, "grad_norm": 0.3707689046859741, "learning_rate": 3.0303030303030305e-07, "loss": 0.0005, "num_tokens": 7287416.0, "reward": 0.9994514584541321, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994514584541321, "reward_meter_std": 2.538687112974003e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.539291381253861e-05, "reward_total_composite_mean": 0.9994514584541321, "reward_total_composite_std": 2.538687112974003e-05, "reward_total_mean": 0.9994514584541321, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994514584541321, "rewards/meter/std": 2.538687112974003e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994514584541321, "rewards/total_composite/std": 2.538687112974003e-05, "sampling/importance_sampling_ratio/max": 1.66619074344635, "sampling/importance_sampling_ratio/mean": 1.0068362951278687, "sampling/importance_sampling_ratio/min": 0.5209580659866333, "sampling/sampling_logp_difference/max": 0.6520857810974121, "sampling/sampling_logp_difference/mean": 0.009306100197136402, "step": 3201 }, { "clip_ratio/high_max": 0.01261215889826417, "clip_ratio/high_mean": 0.01261215889826417, "clip_ratio/low_mean": 0.003929318394511938, "clip_ratio/low_min": 0.003929318394511938, "clip_ratio/region_mean": 0.016541477292776108, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 128.375, "completions/mean_terminated_length": 128.375, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.16156750172376633, "epoch": 0.12860987267542273, "frac_reward_zero_std": 0.0, "grad_norm": 2.368190050125122, "learning_rate": 3.0000000000000004e-07, "loss": -0.0017, "num_tokens": 7289899.0, "reward": 0.962304949760437, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979463815689087, "reward_meter_std": 3.4329503250773996e-05, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.0659857988357544, "reward_total_composite_mean": 0.962304949760437, "reward_total_composite_std": 0.06598580628633499, "reward_total_mean": 0.962304949760437, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979463815689087, "rewards/meter/std": 3.4329503250773996e-05, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.962304949760437, "rewards/total_composite/std": 0.06598580628633499, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0038442611694336, "sampling/importance_sampling_ratio/min": 0.29717305302619934, "sampling/sampling_logp_difference/max": 1.2134406566619873, "sampling/sampling_logp_difference/mean": 0.017233658581972122, "step": 3202 }, { "clip_ratio/high_max": 0.01714910357259214, "clip_ratio/high_mean": 0.01714910357259214, "clip_ratio/low_mean": 0.006470522261224687, "clip_ratio/low_min": 0.006470522261224687, "clip_ratio/region_mean": 0.023619625833816826, "completions/clipped_ratio": 0.0, "completions/max_length": 294.0, "completions/max_terminated_length": 294.0, "completions/mean_length": 281.125, "completions/mean_terminated_length": 281.125, "completions/min_length": 260.0, "completions/min_terminated_length": 260.0, "entropy": 0.34676245227456093, "epoch": 0.12865003815720769, "frac_reward_zero_std": 0.0, "grad_norm": 2.0183863639831543, "learning_rate": 2.96969696969697e-07, "loss": 0.0342, "num_tokens": 7293996.0, "reward": 0.8708112835884094, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.921875, "reward_count_adherence_std": 0.06469365209341049, "reward_meter_mean": 0.9827648401260376, "reward_meter_std": 0.0459895133972168, "reward_repeat_penalty_mean": 0.9612745046615601, "reward_repeat_penalty_std": 0.045027956366539, "reward_std": 0.08316829800605774, "reward_total_composite_mean": 0.8708112835884094, "reward_total_composite_std": 0.08316829800605774, "reward_total_mean": 0.8708112835884094, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.921875, "rewards/count_adherence/std": 0.06469365209341049, "rewards/meter/mean": 0.9827648401260376, "rewards/meter/std": 0.0459895133972168, "rewards/repeat_penalty/mean": 0.9612745046615601, "rewards/repeat_penalty/std": 0.045027956366539, "rewards/total_composite/mean": 0.8708112835884094, "rewards/total_composite/std": 0.08316829800605774, "sampling/importance_sampling_ratio/max": 1.6892542839050293, "sampling/importance_sampling_ratio/mean": 1.0026259422302246, "sampling/importance_sampling_ratio/min": 0.07883427292108536, "sampling/sampling_logp_difference/max": 2.540407419204712, "sampling/sampling_logp_difference/mean": 0.040366996079683304, "step": 3203 }, { "clip_ratio/high_max": 0.0020491802133619785, "clip_ratio/high_mean": 0.0020491802133619785, "clip_ratio/low_mean": 0.0020491802133619785, "clip_ratio/low_min": 0.0020491802133619785, "clip_ratio/region_mean": 0.004098360426723957, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.014257052214816213, "epoch": 0.12869020363899264, "frac_reward_zero_std": 0.0, "grad_norm": 0.02218056656420231, "learning_rate": 2.9393939393939397e-07, "loss": 0.0001, "num_tokens": 7295796.0, "reward": 0.9973360300064087, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973360300064087, "reward_meter_std": 8.113272997434251e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.10424353403505e-06, "reward_total_composite_mean": 0.9973360300064087, "reward_total_composite_std": 8.113272997434251e-06, "reward_total_mean": 0.9973360300064087, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973360300064087, "rewards/meter/std": 8.113272997434251e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973360300064087, "rewards/total_composite/std": 8.113272997434251e-06, "sampling/importance_sampling_ratio/max": 1.2522022724151611, "sampling/importance_sampling_ratio/mean": 1.0012545585632324, "sampling/importance_sampling_ratio/min": 0.8462150692939758, "sampling/sampling_logp_difference/max": 0.22490382194519043, "sampling/sampling_logp_difference/mean": 0.002050558337941766, "step": 3204 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 28.0, "completions/max_terminated_length": 28.0, "completions/mean_length": 28.0, "completions/mean_terminated_length": 28.0, "completions/min_length": 28.0, "completions/min_terminated_length": 28.0, "entropy": 0.0011580012142076157, "epoch": 0.1287303691207776, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.9090909090909096e-07, "loss": 0.0, "num_tokens": 7297276.0, "reward": 0.9943599104881287, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9943599104881287, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9943599104881287, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9943599104881287, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9943599104881287, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9943599104881287, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0014938116073608, "sampling/importance_sampling_ratio/mean": 1.0001215934753418, "sampling/importance_sampling_ratio/min": 0.9986100196838379, "sampling/sampling_logp_difference/max": 0.0014927273150533438, "sampling/sampling_logp_difference/mean": 0.0001377786829834804, "step": 3205 }, { "clip_ratio/high_max": 0.012698987615294755, "clip_ratio/high_mean": 0.012698987615294755, "clip_ratio/low_mean": 0.005936880130320787, "clip_ratio/low_min": 0.005936880130320787, "clip_ratio/region_mean": 0.018635867745615542, "completions/clipped_ratio": 0.0, "completions/max_length": 131.0, "completions/max_terminated_length": 131.0, "completions/mean_length": 127.25, "completions/mean_terminated_length": 127.25, "completions/min_length": 125.0, "completions/min_terminated_length": 125.0, "entropy": 0.31193940714001656, "epoch": 0.12877053460256255, "frac_reward_zero_std": 0.0, "grad_norm": 2.458156108856201, "learning_rate": 2.878787878787879e-07, "loss": 0.0027, "num_tokens": 7299630.0, "reward": 0.9546165466308594, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9723162651062012, "reward_meter_std": 0.04477069899439812, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.06121518462896347, "reward_total_composite_mean": 0.9546165466308594, "reward_total_composite_std": 0.06121518090367317, "reward_total_mean": 0.9546165466308594, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9723162651062012, "rewards/meter/std": 0.04477069899439812, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9546165466308594, "rewards/total_composite/std": 0.06121518090367317, "sampling/importance_sampling_ratio/max": 1.842721700668335, "sampling/importance_sampling_ratio/mean": 1.0102778673171997, "sampling/importance_sampling_ratio/min": 0.2634107768535614, "sampling/sampling_logp_difference/max": 1.334040641784668, "sampling/sampling_logp_difference/mean": 0.033622466027736664, "step": 3206 }, { "clip_ratio/high_max": 0.004061477375216782, "clip_ratio/high_mean": 0.004061477375216782, "clip_ratio/low_mean": 0.0013440860202535987, "clip_ratio/low_min": 0.0013440860202535987, "clip_ratio/region_mean": 0.005405563395470381, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 92.875, "completions/mean_terminated_length": 92.875, "completions/min_length": 92.0, "completions/min_terminated_length": 92.0, "entropy": 0.07594481343403459, "epoch": 0.1288107000843475, "frac_reward_zero_std": 0.0, "grad_norm": 3.696357250213623, "learning_rate": 2.848484848484849e-07, "loss": 0.001, "num_tokens": 7301645.0, "reward": 0.9976005554199219, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976005554199219, "reward_meter_std": 0.0004543311079032719, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00045434184721671045, "reward_total_composite_mean": 0.9976005554199219, "reward_total_composite_std": 0.0004543311079032719, "reward_total_mean": 0.9976005554199219, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976005554199219, "rewards/meter/std": 0.0004543311079032719, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9976005554199219, "rewards/total_composite/std": 0.0004543311079032719, "sampling/importance_sampling_ratio/max": 1.5881195068359375, "sampling/importance_sampling_ratio/mean": 1.0031743049621582, "sampling/importance_sampling_ratio/min": 0.4846150279045105, "sampling/sampling_logp_difference/max": 0.724400520324707, "sampling/sampling_logp_difference/mean": 0.009967589750885963, "step": 3207 }, { "clip_ratio/high_max": 0.0013440860202535987, "clip_ratio/high_mean": 0.0013440860202535987, "clip_ratio/low_mean": 0.006838270695880055, "clip_ratio/low_min": 0.006838270695880055, "clip_ratio/region_mean": 0.008182356716133654, "completions/clipped_ratio": 0.0, "completions/max_length": 93.0, "completions/max_terminated_length": 93.0, "completions/mean_length": 92.625, "completions/mean_terminated_length": 92.625, "completions/min_length": 91.0, "completions/min_terminated_length": 91.0, "entropy": 0.08394959289580584, "epoch": 0.12885086556613246, "frac_reward_zero_std": 0.0, "grad_norm": 2.1962215900421143, "learning_rate": 2.818181818181819e-07, "loss": -0.0039, "num_tokens": 7303658.0, "reward": 0.9975332021713257, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9975332021713257, "reward_meter_std": 0.0004231163766235113, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00042310450226068497, "reward_total_composite_mean": 0.9975332021713257, "reward_total_composite_std": 0.0004231163766235113, "reward_total_mean": 0.9975332021713257, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9975332021713257, "rewards/meter/std": 0.0004231163766235113, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9975332021713257, "rewards/total_composite/std": 0.0004231163766235113, "sampling/importance_sampling_ratio/max": 1.477807641029358, "sampling/importance_sampling_ratio/mean": 1.0051023960113525, "sampling/importance_sampling_ratio/min": 0.38812631368637085, "sampling/sampling_logp_difference/max": 0.9464244842529297, "sampling/sampling_logp_difference/mean": 0.010751576162874699, "step": 3208 }, { "clip_ratio/high_max": 0.00788172468310222, "clip_ratio/high_mean": 0.00788172468310222, "clip_ratio/low_mean": 0.004707215121015906, "clip_ratio/low_min": 0.004707215121015906, "clip_ratio/region_mean": 0.012588939804118127, "completions/clipped_ratio": 0.0, "completions/max_length": 161.0, "completions/max_terminated_length": 161.0, "completions/mean_length": 159.375, "completions/mean_terminated_length": 159.375, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.20060089603066444, "epoch": 0.1288910310479174, "frac_reward_zero_std": 0.0, "grad_norm": 1.3126479387283325, "learning_rate": 2.787878787878788e-07, "loss": 0.0075, "num_tokens": 7306277.0, "reward": 0.9422216415405273, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9976499080657959, "reward_meter_std": 0.0005515830125659704, "reward_repeat_penalty_mean": 0.9444444179534912, "reward_repeat_penalty_std": 0.059391383081674576, "reward_std": 0.059194166213274, "reward_total_composite_mean": 0.9422216415405273, "reward_total_composite_std": 0.05919419974088669, "reward_total_mean": 0.9422216415405273, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9976499080657959, "rewards/meter/std": 0.0005515830125659704, "rewards/repeat_penalty/mean": 0.9444444179534912, "rewards/repeat_penalty/std": 0.059391383081674576, "rewards/total_composite/mean": 0.9422216415405273, "rewards/total_composite/std": 0.05919419974088669, "sampling/importance_sampling_ratio/max": 1.4307658672332764, "sampling/importance_sampling_ratio/mean": 1.0039074420928955, "sampling/importance_sampling_ratio/min": 0.13950084149837494, "sampling/sampling_logp_difference/max": 1.9696846008300781, "sampling/sampling_logp_difference/mean": 0.020425979048013687, "step": 3209 }, { "clip_ratio/high_max": 0.0052327855955809355, "clip_ratio/high_mean": 0.0052327855955809355, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/region_mean": 0.006968896719627082, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.625, "completions/mean_terminated_length": 71.625, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.08111674338579178, "epoch": 0.12893119652970236, "frac_reward_zero_std": 0.0, "grad_norm": 0.4994388222694397, "learning_rate": 2.757575757575758e-07, "loss": -0.0001, "num_tokens": 7308194.0, "reward": 0.9994117021560669, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994117021560669, "reward_meter_std": 7.800360617693514e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.79872716520913e-05, "reward_total_composite_mean": 0.9994117021560669, "reward_total_composite_std": 7.800360617693514e-05, "reward_total_mean": 0.9994117021560669, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994117021560669, "rewards/meter/std": 7.800360617693514e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994117021560669, "rewards/total_composite/std": 7.800360617693514e-05, "sampling/importance_sampling_ratio/max": 1.3537853956222534, "sampling/importance_sampling_ratio/mean": 1.0026097297668457, "sampling/importance_sampling_ratio/min": 0.4058171510696411, "sampling/sampling_logp_difference/max": 0.9018526077270508, "sampling/sampling_logp_difference/mean": 0.010017759166657925, "step": 3210 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 80.0, "completions/max_terminated_length": 80.0, "completions/mean_length": 80.0, "completions/mean_terminated_length": 80.0, "completions/min_length": 80.0, "completions/min_terminated_length": 80.0, "entropy": 0.0023747361556161195, "epoch": 0.12897136201148732, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.7272727272727274e-07, "loss": 0.0, "num_tokens": 7310146.0, "reward": 0.7190229296684265, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.7190229296684265, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.7190229296684265, "reward_total_composite_std": 0.0, "reward_total_mean": 0.7190229296684265, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.7190229296684265, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.7190229296684265, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0070383548736572, "sampling/importance_sampling_ratio/mean": 1.0002856254577637, "sampling/importance_sampling_ratio/min": 0.9983328580856323, "sampling/sampling_logp_difference/max": 0.007013600319623947, "sampling/sampling_logp_difference/mean": 0.00030469015473499894, "step": 3211 }, { "clip_ratio/high_max": 0.009384893695823848, "clip_ratio/high_mean": 0.009384893695823848, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.013172772596590221, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.052677970845252275, "epoch": 0.12901152749327227, "frac_reward_zero_std": 0.0, "grad_norm": 0.7813161611557007, "learning_rate": 2.6969696969696973e-07, "loss": -0.003, "num_tokens": 7311944.0, "reward": 0.998114824295044, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998114824295044, "reward_meter_std": 6.83319303789176e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 6.832612416474149e-05, "reward_total_composite_mean": 0.998114824295044, "reward_total_composite_std": 6.83319303789176e-05, "reward_total_mean": 0.998114824295044, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998114824295044, "rewards/meter/std": 6.83319303789176e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998114824295044, "rewards/total_composite/std": 6.83319303789176e-05, "sampling/importance_sampling_ratio/max": 1.6796454191207886, "sampling/importance_sampling_ratio/mean": 0.9998994469642639, "sampling/importance_sampling_ratio/min": 0.29217007756233215, "sampling/sampling_logp_difference/max": 1.2304191589355469, "sampling/sampling_logp_difference/mean": 0.014992273412644863, "step": 3212 }, { "clip_ratio/high_max": 0.030003347201272845, "clip_ratio/high_mean": 0.030003347201272845, "clip_ratio/low_mean": 0.002218934940174222, "clip_ratio/low_min": 0.002218934940174222, "clip_ratio/region_mean": 0.03222228214144707, "completions/clipped_ratio": 0.0, "completions/max_length": 338.0, "completions/max_terminated_length": 338.0, "completions/mean_length": 315.5, "completions/mean_terminated_length": 315.5, "completions/min_length": 307.0, "completions/min_terminated_length": 307.0, "entropy": 0.4627891331911087, "epoch": 0.12905169297505723, "frac_reward_zero_std": 0.0, "grad_norm": 1.7816143035888672, "learning_rate": 2.666666666666667e-07, "loss": 0.0308, "num_tokens": 7315948.0, "reward": 0.9913522005081177, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991602897644043, "reward_meter_std": 0.00017255292914342135, "reward_repeat_penalty_mean": 0.9921875, "reward_repeat_penalty_std": 0.022097086533904076, "reward_std": 0.021975882351398468, "reward_total_composite_mean": 0.9913522005081177, "reward_total_composite_std": 0.021975873038172722, "reward_total_mean": 0.9913522005081177, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991602897644043, "rewards/meter/std": 0.00017255292914342135, "rewards/repeat_penalty/mean": 0.9921875, "rewards/repeat_penalty/std": 0.022097086533904076, "rewards/total_composite/mean": 0.9913522005081177, "rewards/total_composite/std": 0.021975873038172722, "sampling/importance_sampling_ratio/max": 1.7197778224945068, "sampling/importance_sampling_ratio/mean": 1.0101171731948853, "sampling/importance_sampling_ratio/min": 0.19695636630058289, "sampling/sampling_logp_difference/max": 1.6247730255126953, "sampling/sampling_logp_difference/mean": 0.04883609339594841, "step": 3213 }, { "clip_ratio/high_max": 0.009855842916294932, "clip_ratio/high_mean": 0.009855842916294932, "clip_ratio/low_mean": 0.003487827139906585, "clip_ratio/low_min": 0.003487827139906585, "clip_ratio/region_mean": 0.013343670056201518, "completions/clipped_ratio": 0.0, "completions/max_length": 181.0, "completions/max_terminated_length": 181.0, "completions/mean_length": 178.375, "completions/mean_terminated_length": 178.375, "completions/min_length": 176.0, "completions/min_terminated_length": 176.0, "entropy": 0.2543965280056, "epoch": 0.12909185845684218, "frac_reward_zero_std": 0.0, "grad_norm": 0.841699481010437, "learning_rate": 2.6363636363636366e-07, "loss": 0.0001, "num_tokens": 7318839.0, "reward": 0.9990983009338379, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990983009338379, "reward_meter_std": 0.0001289558131247759, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00012895492545794696, "reward_total_composite_mean": 0.9990983009338379, "reward_total_composite_std": 0.0001289558131247759, "reward_total_mean": 0.9990983009338379, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990983009338379, "rewards/meter/std": 0.0001289558131247759, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990983009338379, "rewards/total_composite/std": 0.0001289558131247759, "sampling/importance_sampling_ratio/max": 1.48737633228302, "sampling/importance_sampling_ratio/mean": 1.0072314739227295, "sampling/importance_sampling_ratio/min": 0.2830124795436859, "sampling/sampling_logp_difference/max": 1.2622642517089844, "sampling/sampling_logp_difference/mean": 0.018457449972629547, "step": 3214 }, { "clip_ratio/high_max": 0.007017801166512072, "clip_ratio/high_mean": 0.007017801166512072, "clip_ratio/low_mean": 0.0035211266949772835, "clip_ratio/low_min": 0.0035211266949772835, "clip_ratio/region_mean": 0.010538927861489356, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.25, "completions/mean_terminated_length": 71.25, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.08346446044743061, "epoch": 0.12913202393862713, "frac_reward_zero_std": 0.0, "grad_norm": 0.2847835123538971, "learning_rate": 2.6060606060606065e-07, "loss": 0.0005, "num_tokens": 7320585.0, "reward": 0.9994255304336548, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994255304336548, "reward_meter_std": 1.9644281564978883e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.9626990251708776e-05, "reward_total_composite_mean": 0.9994255304336548, "reward_total_composite_std": 1.9644281564978883e-05, "reward_total_mean": 0.9994255304336548, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994255304336548, "rewards/meter/std": 1.9644281564978883e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994255304336548, "rewards/total_composite/std": 1.9644281564978883e-05, "sampling/importance_sampling_ratio/max": 1.3071258068084717, "sampling/importance_sampling_ratio/mean": 1.001387357711792, "sampling/importance_sampling_ratio/min": 0.24264176189899445, "sampling/sampling_logp_difference/max": 1.4161691665649414, "sampling/sampling_logp_difference/mean": 0.012419765815138817, "step": 3215 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 29.0, "completions/max_terminated_length": 29.0, "completions/mean_length": 29.0, "completions/mean_terminated_length": 29.0, "completions/min_length": 29.0, "completions/min_terminated_length": 29.0, "entropy": 0.003761589468922466, "epoch": 0.1291721894204121, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.575757575757576e-07, "loss": 0.0, "num_tokens": 7321969.0, "reward": 0.9957436919212341, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957436919212341, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9957436919212341, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9957436919212341, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957436919212341, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9957436919212341, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0145117044448853, "sampling/importance_sampling_ratio/mean": 1.0002456903457642, "sampling/importance_sampling_ratio/min": 0.9782066345214844, "sampling/sampling_logp_difference/max": 0.022034302353858948, "sampling/sampling_logp_difference/mean": 0.00043711578473448753, "step": 3216 }, { "clip_ratio/high_max": 0.0012755101779475808, "clip_ratio/high_mean": 0.0012755101779475808, "clip_ratio/low_mean": 0.0012755101779475808, "clip_ratio/low_min": 0.0012755101779475808, "clip_ratio/region_mean": 0.0025510203558951616, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.035912956576794386, "epoch": 0.12921235490219704, "frac_reward_zero_std": 0.0, "grad_norm": 0.06737548857927322, "learning_rate": 2.545454545454546e-07, "loss": 0.0004, "num_tokens": 7324185.0, "reward": 0.9994082450866699, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994082450866699, "reward_meter_std": 3.3991077543760184e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.404888502700487e-06, "reward_total_composite_mean": 0.9994082450866699, "reward_total_composite_std": 3.3991077543760184e-06, "reward_total_mean": 0.9994082450866699, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994082450866699, "rewards/meter/std": 3.3991077543760184e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994082450866699, "rewards/total_composite/std": 3.3991077543760184e-06, "sampling/importance_sampling_ratio/max": 1.321240782737732, "sampling/importance_sampling_ratio/mean": 1.0018640756607056, "sampling/importance_sampling_ratio/min": 0.689630925655365, "sampling/sampling_logp_difference/max": 0.3715987205505371, "sampling/sampling_logp_difference/mean": 0.003399990499019623, "step": 3217 }, { "clip_ratio/high_max": 0.0061869441997259855, "clip_ratio/high_mean": 0.0061869441997259855, "clip_ratio/low_mean": 0.009702177019789815, "clip_ratio/low_min": 0.009702177019789815, "clip_ratio/region_mean": 0.0158891212195158, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 141.375, "completions/mean_terminated_length": 141.375, "completions/min_length": 139.0, "completions/min_terminated_length": 139.0, "entropy": 0.21374017372727394, "epoch": 0.129252520383982, "frac_reward_zero_std": 0.0, "grad_norm": 1.3049845695495605, "learning_rate": 2.515151515151515e-07, "loss": -0.0009, "num_tokens": 7326588.0, "reward": 0.9991159439086914, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991159439086914, "reward_meter_std": 0.00015829414769541472, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00015829637413844466, "reward_total_composite_mean": 0.9991159439086914, "reward_total_composite_std": 0.00015829414769541472, "reward_total_mean": 0.9991159439086914, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991159439086914, "rewards/meter/std": 0.00015829414769541472, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991159439086914, "rewards/total_composite/std": 0.00015829414769541472, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0080312490463257, "sampling/importance_sampling_ratio/min": 0.40840601921081543, "sampling/sampling_logp_difference/max": 0.8954935073852539, "sampling/sampling_logp_difference/mean": 0.020292429253458977, "step": 3218 }, { "clip_ratio/high_max": 0.017989110434427857, "clip_ratio/high_mean": 0.017989110434427857, "clip_ratio/low_mean": 0.0102413734421134, "clip_ratio/low_min": 0.0102413734421134, "clip_ratio/region_mean": 0.028230483876541257, "completions/clipped_ratio": 0.0, "completions/max_length": 111.0, "completions/max_terminated_length": 111.0, "completions/mean_length": 105.125, "completions/mean_terminated_length": 105.125, "completions/min_length": 102.0, "completions/min_terminated_length": 102.0, "entropy": 0.3442672826349735, "epoch": 0.12929268586576695, "frac_reward_zero_std": 0.0, "grad_norm": 5.071619510650635, "learning_rate": 2.484848484848485e-07, "loss": 0.0279, "num_tokens": 7328757.0, "reward": 0.9915302991867065, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9915302991867065, "reward_meter_std": 0.006453692447394133, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006453694310039282, "reward_total_composite_mean": 0.9915302991867065, "reward_total_composite_std": 0.006453692447394133, "reward_total_mean": 0.9915302991867065, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9915302991867065, "rewards/meter/std": 0.006453692447394133, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9915302991867065, "rewards/total_composite/std": 0.006453692447394133, "sampling/importance_sampling_ratio/max": 1.9355528354644775, "sampling/importance_sampling_ratio/mean": 1.0104427337646484, "sampling/importance_sampling_ratio/min": 0.4566134214401245, "sampling/sampling_logp_difference/max": 0.7839181423187256, "sampling/sampling_logp_difference/mean": 0.0352078415453434, "step": 3219 }, { "clip_ratio/high_max": 0.014640397042967379, "clip_ratio/high_mean": 0.014640397042967379, "clip_ratio/low_mean": 0.005359299597330391, "clip_ratio/low_min": 0.005359299597330391, "clip_ratio/region_mean": 0.01999969664029777, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 68.625, "completions/mean_terminated_length": 68.625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.21459667198359966, "epoch": 0.1293328513475519, "frac_reward_zero_std": 0.0, "grad_norm": 3.3606433868408203, "learning_rate": 2.4545454545454545e-07, "loss": 0.0085, "num_tokens": 7330682.0, "reward": 0.9950666427612305, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9950666427612305, "reward_meter_std": 0.0027368878945708275, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0027369032613933086, "reward_total_composite_mean": 0.9950666427612305, "reward_total_composite_std": 0.0027368878945708275, "reward_total_mean": 0.9950666427612305, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9950666427612305, "rewards/meter/std": 0.0027368878945708275, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9950666427612305, "rewards/total_composite/std": 0.0027368878945708275, "sampling/importance_sampling_ratio/max": 1.609100103378296, "sampling/importance_sampling_ratio/mean": 1.0102657079696655, "sampling/importance_sampling_ratio/min": 0.45933452248573303, "sampling/sampling_logp_difference/max": 0.7779765129089355, "sampling/sampling_logp_difference/mean": 0.024690676480531693, "step": 3220 }, { "clip_ratio/high_max": 0.012232337845489383, "clip_ratio/high_mean": 0.012232337845489383, "clip_ratio/low_mean": 0.024722653324715793, "clip_ratio/low_min": 0.024722653324715793, "clip_ratio/region_mean": 0.036954991170205176, "completions/clipped_ratio": 0.0, "completions/max_length": 75.0, "completions/max_terminated_length": 75.0, "completions/mean_length": 71.0, "completions/mean_terminated_length": 71.0, "completions/min_length": 69.0, "completions/min_terminated_length": 69.0, "entropy": 0.2918657101690769, "epoch": 0.12937301682933686, "frac_reward_zero_std": 0.0, "grad_norm": 3.8582441806793213, "learning_rate": 2.4242424242424244e-07, "loss": 0.0021, "num_tokens": 7332402.0, "reward": 0.9949785470962524, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9949785470962524, "reward_meter_std": 0.0019115530885756016, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0019115599570795894, "reward_total_composite_mean": 0.9949785470962524, "reward_total_composite_std": 0.0019115530885756016, "reward_total_mean": 0.9949785470962524, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9949785470962524, "rewards/meter/std": 0.0019115530885756016, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9949785470962524, "rewards/total_composite/std": 0.0019115530885756016, "sampling/importance_sampling_ratio/max": 1.7042900323867798, "sampling/importance_sampling_ratio/mean": 1.0079292058944702, "sampling/importance_sampling_ratio/min": 0.24765028059482574, "sampling/sampling_logp_difference/max": 1.395737648010254, "sampling/sampling_logp_difference/mean": 0.03391466289758682, "step": 3221 }, { "clip_ratio/high_max": 0.01708377245813608, "clip_ratio/high_mean": 0.01708377245813608, "clip_ratio/low_mean": 0.018319292226806283, "clip_ratio/low_min": 0.018319292226806283, "clip_ratio/region_mean": 0.035403064684942365, "completions/clipped_ratio": 0.0, "completions/max_length": 205.0, "completions/max_terminated_length": 205.0, "completions/mean_length": 197.75, "completions/mean_terminated_length": 197.75, "completions/min_length": 194.0, "completions/min_terminated_length": 194.0, "entropy": 0.36750227957963943, "epoch": 0.1294131823111218, "frac_reward_zero_std": 0.0, "grad_norm": 2.1812803745269775, "learning_rate": 2.3939393939393943e-07, "loss": 0.0076, "num_tokens": 7335408.0, "reward": 0.9989155530929565, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989155530929565, "reward_meter_std": 0.00026041135424748063, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002604102483019233, "reward_total_composite_mean": 0.9989155530929565, "reward_total_composite_std": 0.00026041135424748063, "reward_total_mean": 0.9989155530929565, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989155530929565, "rewards/meter/std": 0.00026041135424748063, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989155530929565, "rewards/total_composite/std": 0.00026041135424748063, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0105738639831543, "sampling/importance_sampling_ratio/min": 0.17348448932170868, "sampling/sampling_logp_difference/max": 1.7516670227050781, "sampling/sampling_logp_difference/mean": 0.04675072431564331, "step": 3222 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0017184649041155353, "epoch": 0.12945334779290676, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.3636363636363637e-07, "loss": 0.0, "num_tokens": 7336960.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0026967525482178, "sampling/importance_sampling_ratio/mean": 1.0001906156539917, "sampling/importance_sampling_ratio/min": 0.9931014776229858, "sampling/sampling_logp_difference/max": 0.006922472268342972, "sampling/sampling_logp_difference/mean": 0.00022982771042734385, "step": 3223 }, { "clip_ratio/high_max": 0.006410256493836641, "clip_ratio/high_mean": 0.006410256493836641, "clip_ratio/low_mean": 0.004746835213154554, "clip_ratio/low_min": 0.004746835213154554, "clip_ratio/region_mean": 0.011157091706991196, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 78.125, "completions/mean_terminated_length": 78.125, "completions/min_length": 77.0, "completions/min_terminated_length": 77.0, "entropy": 0.1890299841761589, "epoch": 0.12949351327469172, "frac_reward_zero_std": 0.0, "grad_norm": 2.2346813678741455, "learning_rate": 2.3333333333333336e-07, "loss": -0.0014, "num_tokens": 7339001.0, "reward": 0.9987478256225586, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987478256225586, "reward_meter_std": 0.000937741482630372, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0009377306560054421, "reward_total_composite_mean": 0.9987478256225586, "reward_total_composite_std": 0.000937741482630372, "reward_total_mean": 0.9987478256225586, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987478256225586, "rewards/meter/std": 0.000937741482630372, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987478256225586, "rewards/total_composite/std": 0.000937741482630372, "sampling/importance_sampling_ratio/max": 1.5145747661590576, "sampling/importance_sampling_ratio/mean": 1.0025646686553955, "sampling/importance_sampling_ratio/min": 0.4738627076148987, "sampling/sampling_logp_difference/max": 0.7468376159667969, "sampling/sampling_logp_difference/mean": 0.016266027465462685, "step": 3224 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0005534949777938891, "epoch": 0.12953367875647667, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.3030303030303032e-07, "loss": 0.0, "num_tokens": 7340889.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0014703273773193, "sampling/importance_sampling_ratio/mean": 1.0000407695770264, "sampling/importance_sampling_ratio/min": 0.9948374629020691, "sampling/sampling_logp_difference/max": 0.005175924859941006, "sampling/sampling_logp_difference/mean": 6.631435098825023e-05, "step": 3225 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.019021739484742284, "clip_ratio/low_min": 0.019021739484742284, "clip_ratio/region_mean": 0.019021739484742284, "completions/clipped_ratio": 0.0, "completions/max_length": 47.0, "completions/max_terminated_length": 47.0, "completions/mean_length": 46.125, "completions/mean_terminated_length": 46.125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.09778465609997511, "epoch": 0.12957384423826163, "frac_reward_zero_std": 0.0, "grad_norm": 5.847517967224121, "learning_rate": 2.2727272727272729e-07, "loss": 0.0043, "num_tokens": 7342490.0, "reward": 0.9431781768798828, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9431781768798828, "reward_meter_std": 0.0021464419551193714, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002146440092474222, "reward_total_composite_mean": 0.9431781768798828, "reward_total_composite_std": 0.0021464419551193714, "reward_total_mean": 0.9431781768798828, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9431781768798828, "rewards/meter/std": 0.0021464419551193714, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9431781768798828, "rewards/total_composite/std": 0.0021464419551193714, "sampling/importance_sampling_ratio/max": 1.9484820365905762, "sampling/importance_sampling_ratio/mean": 1.005907416343689, "sampling/importance_sampling_ratio/min": 0.31117722392082214, "sampling/sampling_logp_difference/max": 1.1673927307128906, "sampling/sampling_logp_difference/mean": 0.020598582923412323, "step": 3226 }, { "clip_ratio/high_max": 0.0037313431967049837, "clip_ratio/high_mean": 0.0037313431967049837, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.005597014795057476, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.047321764286607504, "epoch": 0.12961400972004658, "frac_reward_zero_std": 0.0, "grad_norm": 0.25633835792541504, "learning_rate": 2.2424242424242425e-07, "loss": -0.0002, "num_tokens": 7344170.0, "reward": 0.9981385469436646, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981385469436646, "reward_meter_std": 1.2708615031442605e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.2711200724879745e-05, "reward_total_composite_mean": 0.9981385469436646, "reward_total_composite_std": 1.2708615031442605e-05, "reward_total_mean": 0.9981385469436646, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981385469436646, "rewards/meter/std": 1.2708615031442605e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981385469436646, "rewards/total_composite/std": 1.2708615031442605e-05, "sampling/importance_sampling_ratio/max": 1.997063398361206, "sampling/importance_sampling_ratio/mean": 1.002211332321167, "sampling/importance_sampling_ratio/min": 0.6211740374565125, "sampling/sampling_logp_difference/max": 0.6916778087615967, "sampling/sampling_logp_difference/mean": 0.007955299690365791, "step": 3227 }, { "clip_ratio/high_max": 0.00815217406488955, "clip_ratio/high_mean": 0.00815217406488955, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/region_mean": 0.010811748448759317, "completions/clipped_ratio": 0.0, "completions/max_length": 47.0, "completions/max_terminated_length": 47.0, "completions/mean_length": 46.125, "completions/mean_terminated_length": 46.125, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.07639594282954931, "epoch": 0.12965417520183153, "frac_reward_zero_std": 0.0, "grad_norm": 8.390018463134766, "learning_rate": 2.2121212121212124e-07, "loss": 0.0145, "num_tokens": 7345827.0, "reward": 0.9406979084014893, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9406979084014893, "reward_meter_std": 0.006258478853851557, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0062584723345935345, "reward_total_composite_mean": 0.9406979084014893, "reward_total_composite_std": 0.006258478853851557, "reward_total_mean": 0.9406979084014893, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9406979084014893, "rewards/meter/std": 0.006258478853851557, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9406979084014893, "rewards/total_composite/std": 0.006258478853851557, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 0.996314287185669, "sampling/importance_sampling_ratio/min": 0.08505111932754517, "sampling/sampling_logp_difference/max": 2.4645028114318848, "sampling/sampling_logp_difference/mean": 0.02605927735567093, "step": 3228 }, { "clip_ratio/high_max": 0.005102040828205645, "clip_ratio/high_mean": 0.005102040828205645, "clip_ratio/low_mean": 0.008941720938310027, "clip_ratio/low_min": 0.008941720938310027, "clip_ratio/region_mean": 0.014043761766515672, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.875, "completions/mean_terminated_length": 97.875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.10427160188555717, "epoch": 0.1296943406836165, "frac_reward_zero_std": 0.0, "grad_norm": 0.8496602773666382, "learning_rate": 2.181818181818182e-07, "loss": -0.0021, "num_tokens": 7348082.0, "reward": 0.9979760646820068, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9979760646820068, "reward_meter_std": 7.630670006619766e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.630588515894488e-05, "reward_total_composite_mean": 0.9979760646820068, "reward_total_composite_std": 7.630670006619766e-05, "reward_total_mean": 0.9979760646820068, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9979760646820068, "rewards/meter/std": 7.630670006619766e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9979760646820068, "rewards/total_composite/std": 7.630670006619766e-05, "sampling/importance_sampling_ratio/max": 1.3757518529891968, "sampling/importance_sampling_ratio/mean": 1.0022556781768799, "sampling/importance_sampling_ratio/min": 0.46612924337387085, "sampling/sampling_logp_difference/max": 0.7632923126220703, "sampling/sampling_logp_difference/mean": 0.01204567588865757, "step": 3229 }, { "clip_ratio/high_max": 0.003759611048735678, "clip_ratio/high_mean": 0.003759611048735678, "clip_ratio/low_mean": 0.0018656715983524919, "clip_ratio/low_min": 0.0018656715983524919, "clip_ratio/region_mean": 0.00562528264708817, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.75, "completions/mean_terminated_length": 66.75, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.04502238845452666, "epoch": 0.12973450616540144, "frac_reward_zero_std": 0.0, "grad_norm": 0.018652301281690598, "learning_rate": 2.1515151515151517e-07, "loss": -0.0006, "num_tokens": 7349856.0, "reward": 0.9981513023376465, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981513023376465, "reward_meter_std": 1.5048592558741802e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.5035095657367492e-06, "reward_total_composite_mean": 0.9981513023376465, "reward_total_composite_std": 1.5048592558741802e-06, "reward_total_mean": 0.9981513023376465, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981513023376465, "rewards/meter/std": 1.5048592558741802e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981513023376465, "rewards/total_composite/std": 1.5048592558741802e-06, "sampling/importance_sampling_ratio/max": 1.4481335878372192, "sampling/importance_sampling_ratio/mean": 1.0017226934432983, "sampling/importance_sampling_ratio/min": 0.5867893099784851, "sampling/sampling_logp_difference/max": 0.5330893993377686, "sampling/sampling_logp_difference/mean": 0.006907826755195856, "step": 3230 }, { "clip_ratio/high_max": 0.006999526987783611, "clip_ratio/high_mean": 0.006999526987783611, "clip_ratio/low_mean": 0.004716981202363968, "clip_ratio/low_min": 0.004716981202363968, "clip_ratio/region_mean": 0.011716508190147579, "completions/clipped_ratio": 0.0, "completions/max_length": 110.0, "completions/max_terminated_length": 110.0, "completions/mean_length": 107.0, "completions/mean_terminated_length": 107.0, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.15556516498327255, "epoch": 0.1297746716471864, "frac_reward_zero_std": 0.0, "grad_norm": 1.449289321899414, "learning_rate": 2.1212121212121216e-07, "loss": -0.0038, "num_tokens": 7352000.0, "reward": 0.9992383718490601, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992383718490601, "reward_meter_std": 0.00011975476081715897, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001197509845951572, "reward_total_composite_mean": 0.9992383718490601, "reward_total_composite_std": 0.00011975476081715897, "reward_total_mean": 0.9992383718490601, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992383718490601, "rewards/meter/std": 0.00011975476081715897, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992383718490601, "rewards/total_composite/std": 0.00011975476081715897, "sampling/importance_sampling_ratio/max": 1.8299890756607056, "sampling/importance_sampling_ratio/mean": 1.0047478675842285, "sampling/importance_sampling_ratio/min": 0.3795926868915558, "sampling/sampling_logp_difference/max": 0.9686565399169922, "sampling/sampling_logp_difference/mean": 0.015148649923503399, "step": 3231 }, { "clip_ratio/high_max": 0.009743589907884598, "clip_ratio/high_mean": 0.009743589907884598, "clip_ratio/low_mean": 0.0032681134762242436, "clip_ratio/low_min": 0.0032681134762242436, "clip_ratio/region_mean": 0.013011703384108841, "completions/clipped_ratio": 0.0, "completions/max_length": 79.0, "completions/max_terminated_length": 79.0, "completions/mean_length": 77.5, "completions/mean_terminated_length": 77.5, "completions/min_length": 75.0, "completions/min_terminated_length": 75.0, "entropy": 0.19562415033578873, "epoch": 0.12981483712897135, "frac_reward_zero_std": 0.0, "grad_norm": 2.6132781505584717, "learning_rate": 2.090909090909091e-07, "loss": -0.0004, "num_tokens": 7354036.0, "reward": 0.9991195797920227, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991195797920227, "reward_meter_std": 0.0001867539540398866, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00018676480976864696, "reward_total_composite_mean": 0.9991195797920227, "reward_total_composite_std": 0.0001867539540398866, "reward_total_mean": 0.9991195797920227, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991195797920227, "rewards/meter/std": 0.0001867539540398866, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991195797920227, "rewards/total_composite/std": 0.0001867539540398866, "sampling/importance_sampling_ratio/max": 1.5898345708847046, "sampling/importance_sampling_ratio/mean": 1.005802035331726, "sampling/importance_sampling_ratio/min": 0.39664843678474426, "sampling/sampling_logp_difference/max": 0.9247050285339355, "sampling/sampling_logp_difference/mean": 0.022667229175567627, "step": 3232 }, { "clip_ratio/high_max": 0.00582152372226119, "clip_ratio/high_mean": 0.00582152372226119, "clip_ratio/low_mean": 0.0009842519648373127, "clip_ratio/low_min": 0.0009842519648373127, "clip_ratio/region_mean": 0.006805775687098503, "completions/clipped_ratio": 0.0, "completions/max_length": 129.0, "completions/max_terminated_length": 129.0, "completions/mean_length": 128.625, "completions/mean_terminated_length": 128.625, "completions/min_length": 127.0, "completions/min_terminated_length": 127.0, "entropy": 0.15214181691408157, "epoch": 0.1298550026107563, "frac_reward_zero_std": 0.0, "grad_norm": 3.281510591506958, "learning_rate": 2.060606060606061e-07, "loss": -0.0061, "num_tokens": 7356649.0, "reward": 0.9830908179283142, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9830908179283142, "reward_meter_std": 0.04200791195034981, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.042007915675640106, "reward_total_composite_mean": 0.9830908179283142, "reward_total_composite_std": 0.04200791195034981, "reward_total_mean": 0.9830908179283142, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9830908179283142, "rewards/meter/std": 0.04200791195034981, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9830908179283142, "rewards/total_composite/std": 0.04200791195034981, "sampling/importance_sampling_ratio/max": 1.564560890197754, "sampling/importance_sampling_ratio/mean": 1.00603187084198, "sampling/importance_sampling_ratio/min": 0.4094855487346649, "sampling/sampling_logp_difference/max": 0.8928537368774414, "sampling/sampling_logp_difference/mean": 0.016228584572672844, "step": 3233 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.001778529680450447, "epoch": 0.12989516809254126, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.0303030303030303e-07, "loss": 0.0, "num_tokens": 7358433.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0019910335540771, "sampling/importance_sampling_ratio/mean": 1.0002152919769287, "sampling/importance_sampling_ratio/min": 0.9995585680007935, "sampling/sampling_logp_difference/max": 0.0019890288822352886, "sampling/sampling_logp_difference/mean": 0.0002218023146269843, "step": 3234 }, { "clip_ratio/high_max": 0.025164754129946232, "clip_ratio/high_mean": 0.025164754129946232, "clip_ratio/low_mean": 0.001879699295386672, "clip_ratio/low_min": 0.001879699295386672, "clip_ratio/region_mean": 0.027044453425332904, "completions/clipped_ratio": 0.0, "completions/max_length": 137.0, "completions/max_terminated_length": 137.0, "completions/mean_length": 133.625, "completions/mean_terminated_length": 133.625, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.2376344669610262, "epoch": 0.1299353335743262, "frac_reward_zero_std": 0.0, "grad_norm": 2.066392660140991, "learning_rate": 2.0000000000000002e-07, "loss": 0.0024, "num_tokens": 7361110.0, "reward": 0.9813442230224609, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991873502731323, "reward_meter_std": 0.00011400806397432461, "reward_repeat_penalty_mean": 0.9821428656578064, "reward_repeat_penalty_std": 0.05050762742757797, "reward_std": 0.05045590177178383, "reward_total_composite_mean": 0.9813442230224609, "reward_total_composite_std": 0.05045589804649353, "reward_total_mean": 0.9813442230224609, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991873502731323, "rewards/meter/std": 0.00011400806397432461, "rewards/repeat_penalty/mean": 0.9821428656578064, "rewards/repeat_penalty/std": 0.05050762742757797, "rewards/total_composite/mean": 0.9813442230224609, "rewards/total_composite/std": 0.05045589804649353, "sampling/importance_sampling_ratio/max": 1.8417998552322388, "sampling/importance_sampling_ratio/mean": 1.0075880289077759, "sampling/importance_sampling_ratio/min": 0.10731999576091766, "sampling/sampling_logp_difference/max": 2.231940269470215, "sampling/sampling_logp_difference/mean": 0.031857188791036606, "step": 3235 }, { "clip_ratio/high_max": 0.02566183707676828, "clip_ratio/high_mean": 0.02566183707676828, "clip_ratio/low_mean": 0.002595155732706189, "clip_ratio/low_min": 0.002595155732706189, "clip_ratio/region_mean": 0.028256992809474468, "completions/clipped_ratio": 0.0, "completions/max_length": 298.0, "completions/max_terminated_length": 298.0, "completions/mean_length": 292.5, "completions/mean_terminated_length": 292.5, "completions/min_length": 278.0, "completions/min_terminated_length": 278.0, "entropy": 0.3465934246778488, "epoch": 0.12997549905611117, "frac_reward_zero_std": 0.0, "grad_norm": 2.294574022293091, "learning_rate": 1.9696969696969698e-07, "loss": -0.0018, "num_tokens": 7365186.0, "reward": 0.6899652481079102, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9674590826034546, "reward_meter_std": 0.049329616129398346, "reward_repeat_penalty_mean": 0.9440359473228455, "reward_repeat_penalty_std": 0.042011942714452744, "reward_std": 0.280244380235672, "reward_total_composite_mean": 0.6899652481079102, "reward_total_composite_std": 0.280244380235672, "reward_total_mean": 0.6899652481079102, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9674590826034546, "rewards/meter/std": 0.049329616129398346, "rewards/repeat_penalty/mean": 0.9440359473228455, "rewards/repeat_penalty/std": 0.042011942714452744, "rewards/total_composite/mean": 0.6899652481079102, "rewards/total_composite/std": 0.280244380235672, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0074106454849243, "sampling/importance_sampling_ratio/min": 0.08277346938848495, "sampling/sampling_logp_difference/max": 2.491647720336914, "sampling/sampling_logp_difference/mean": 0.03874696418642998, "step": 3236 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0005204225053603295, "epoch": 0.13001566453789612, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.9393939393939395e-07, "loss": 0.0, "num_tokens": 7366962.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0009338855743408, "sampling/importance_sampling_ratio/mean": 1.0000455379486084, "sampling/importance_sampling_ratio/min": 0.9994771480560303, "sampling/sampling_logp_difference/max": 0.0009334206115454435, "sampling/sampling_logp_difference/mean": 5.0931172154378146e-05, "step": 3237 }, { "clip_ratio/high_max": 0.011160198540892452, "clip_ratio/high_mean": 0.011160198540892452, "clip_ratio/low_mean": 0.007562352577224374, "clip_ratio/low_min": 0.007562352577224374, "clip_ratio/region_mean": 0.018722551118116826, "completions/clipped_ratio": 0.0, "completions/max_length": 138.0, "completions/max_terminated_length": 138.0, "completions/mean_length": 133.75, "completions/mean_terminated_length": 133.75, "completions/min_length": 129.0, "completions/min_terminated_length": 129.0, "entropy": 0.22379975207149982, "epoch": 0.13005583001968107, "frac_reward_zero_std": 0.0, "grad_norm": 4.2172322273254395, "learning_rate": 1.9090909090909094e-07, "loss": -0.0084, "num_tokens": 7369424.0, "reward": 0.9989288449287415, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989288449287415, "reward_meter_std": 0.0006227882695384324, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006227805861271918, "reward_total_composite_mean": 0.9989288449287415, "reward_total_composite_std": 0.0006227882695384324, "reward_total_mean": 0.9989288449287415, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989288449287415, "rewards/meter/std": 0.0006227882695384324, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9989288449287415, "rewards/total_composite/std": 0.0006227882695384324, "sampling/importance_sampling_ratio/max": 1.8821479082107544, "sampling/importance_sampling_ratio/mean": 1.0042328834533691, "sampling/importance_sampling_ratio/min": 0.0024829350877553225, "sampling/sampling_logp_difference/max": 5.998313903808594, "sampling/sampling_logp_difference/mean": 0.035977549850940704, "step": 3238 }, { "clip_ratio/high_max": 0.0018382353009656072, "clip_ratio/high_mean": 0.0018382353009656072, "clip_ratio/low_mean": 0.0018382353009656072, "clip_ratio/low_min": 0.0018382353009656072, "clip_ratio/region_mean": 0.0036764706019312143, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 68.0, "completions/min_terminated_length": 68.0, "entropy": 0.04902258049696684, "epoch": 0.13009599550146603, "frac_reward_zero_std": 0.0, "grad_norm": 1.0152816772460938, "learning_rate": 1.878787878787879e-07, "loss": -0.0012, "num_tokens": 7371177.0, "reward": 0.999480128288269, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999480128288269, "reward_meter_std": 7.362089672824368e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 7.363814802374691e-05, "reward_total_composite_mean": 0.999480128288269, "reward_total_composite_std": 7.362089672824368e-05, "reward_total_mean": 0.999480128288269, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999480128288269, "rewards/meter/std": 7.362089672824368e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999480128288269, "rewards/total_composite/std": 7.362089672824368e-05, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0008584260940552, "sampling/importance_sampling_ratio/min": 0.2629310190677643, "sampling/sampling_logp_difference/max": 1.374150037765503, "sampling/sampling_logp_difference/mean": 0.011260369792580605, "step": 3239 }, { "clip_ratio/high_max": 0.024295849725604057, "clip_ratio/high_mean": 0.024295849725604057, "clip_ratio/low_mean": 0.005920994793996215, "clip_ratio/low_min": 0.005920994793996215, "clip_ratio/region_mean": 0.030216844519600272, "completions/clipped_ratio": 0.0, "completions/max_length": 380.0, "completions/max_terminated_length": 380.0, "completions/mean_length": 350.75, "completions/mean_terminated_length": 350.75, "completions/min_length": 333.0, "completions/min_terminated_length": 333.0, "entropy": 0.49749621003866196, "epoch": 0.13013616098325098, "frac_reward_zero_std": 0.0, "grad_norm": 1.8787816762924194, "learning_rate": 1.8484848484848486e-07, "loss": -0.0318, "num_tokens": 7375719.0, "reward": 0.8923759460449219, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8999999761581421, "reward_count_adherence_std": 0.05345224589109421, "reward_meter_mean": 0.9988770484924316, "reward_meter_std": 0.0009327766601927578, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_std": 0.056543245911598206, "reward_total_composite_mean": 0.8923759460449219, "reward_total_composite_std": 0.056543249636888504, "reward_total_mean": 0.8923759460449219, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8999999761581421, "rewards/count_adherence/std": 0.05345224589109421, "rewards/meter/mean": 0.9988770484924316, "rewards/meter/std": 0.0009327766601927578, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.8923759460449219, "rewards/total_composite/std": 0.056543249636888504, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0113011598587036, "sampling/importance_sampling_ratio/min": 0.008883286267518997, "sampling/sampling_logp_difference/max": 4.723583698272705, "sampling/sampling_logp_difference/mean": 0.05698416754603386, "step": 3240 }, { "clip_ratio/high_max": 0.004464285913854837, "clip_ratio/high_mean": 0.004464285913854837, "clip_ratio/low_mean": 0.019117801217362285, "clip_ratio/low_min": 0.019117801217362285, "clip_ratio/region_mean": 0.023582087131217122, "completions/clipped_ratio": 0.0, "completions/max_length": 392.0, "completions/max_terminated_length": 392.0, "completions/mean_length": 353.125, "completions/mean_terminated_length": 353.125, "completions/min_length": 342.0, "completions/min_terminated_length": 342.0, "entropy": 0.4680623523890972, "epoch": 0.13017632646503594, "frac_reward_zero_std": 0.0, "grad_norm": 1.581771969795227, "learning_rate": 1.8181818181818183e-07, "loss": -0.0301, "num_tokens": 7380120.0, "reward": 0.754129946231842, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7604166269302368, "reward_count_adherence_std": 0.029462775215506554, "reward_meter_mean": 0.9989774823188782, "reward_meter_std": 0.0006195669411681592, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_std": 0.03522040322422981, "reward_total_composite_mean": 0.754129946231842, "reward_total_composite_std": 0.03522041067481041, "reward_total_mean": 0.754129946231842, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7604166269302368, "rewards/count_adherence/std": 0.029462775215506554, "rewards/meter/mean": 0.9989774823188782, "rewards/meter/std": 0.0006195669411681592, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.754129946231842, "rewards/total_composite/std": 0.03522041067481041, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0124188661575317, "sampling/importance_sampling_ratio/min": 0.1727781593799591, "sampling/sampling_logp_difference/max": 1.755746841430664, "sampling/sampling_logp_difference/mean": 0.048592861741781235, "step": 3241 }, { "clip_ratio/high_max": 0.013120089075528085, "clip_ratio/high_mean": 0.013120089075528085, "clip_ratio/low_mean": 0.012771393987350166, "clip_ratio/low_min": 0.012771393987350166, "clip_ratio/region_mean": 0.02589148306287825, "completions/clipped_ratio": 0.0, "completions/max_length": 283.0, "completions/max_terminated_length": 283.0, "completions/mean_length": 275.375, "completions/mean_terminated_length": 275.375, "completions/min_length": 269.0, "completions/min_terminated_length": 269.0, "entropy": 0.39379945397377014, "epoch": 0.1302164919468209, "frac_reward_zero_std": 0.0, "grad_norm": 2.48710560798645, "learning_rate": 1.7878787878787882e-07, "loss": 0.015, "num_tokens": 7383827.0, "reward": 0.8207110166549683, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9965729713439941, "reward_meter_std": 0.0008044191054068506, "reward_repeat_penalty_mean": 0.9411764740943909, "reward_repeat_penalty_std": 0.03144249692559242, "reward_std": 0.027550052851438522, "reward_total_composite_mean": 0.8207110166549683, "reward_total_composite_std": 0.027550045400857925, "reward_total_mean": 0.8207110166549683, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9965729713439941, "rewards/meter/std": 0.0008044191054068506, "rewards/repeat_penalty/mean": 0.9411764740943909, "rewards/repeat_penalty/std": 0.03144249692559242, "rewards/total_composite/mean": 0.8207110166549683, "rewards/total_composite/std": 0.027550045400857925, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0088013410568237, "sampling/importance_sampling_ratio/min": 0.2739553153514862, "sampling/sampling_logp_difference/max": 1.294790267944336, "sampling/sampling_logp_difference/mean": 0.04147728160023689, "step": 3242 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 158.0, "completions/max_terminated_length": 158.0, "completions/mean_length": 158.0, "completions/mean_terminated_length": 158.0, "completions/min_length": 158.0, "completions/min_terminated_length": 158.0, "entropy": 0.009368695667944849, "epoch": 0.13025665742860584, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.7575757575757576e-07, "loss": 0.0, "num_tokens": 7386683.0, "reward": 0.47737064957618713, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.6563846468925476, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 0.7272727489471436, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.47737064957618713, "reward_total_composite_std": 0.0, "reward_total_mean": 0.47737064957618713, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.6563846468925476, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 0.7272727489471436, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.47737064957618713, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0544536113739014, "sampling/importance_sampling_ratio/mean": 1.0010987520217896, "sampling/importance_sampling_ratio/min": 0.9712576866149902, "sampling/sampling_logp_difference/max": 0.05302273482084274, "sampling/sampling_logp_difference/mean": 0.0011825277470052242, "step": 3243 }, { "clip_ratio/high_max": 0.02562888152897358, "clip_ratio/high_mean": 0.02562888152897358, "clip_ratio/low_mean": 0.003246753243729472, "clip_ratio/low_min": 0.003246753243729472, "clip_ratio/region_mean": 0.02887563477270305, "completions/clipped_ratio": 0.0, "completions/max_length": 159.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 155.625, "completions/mean_terminated_length": 155.625, "completions/min_length": 153.0, "completions/min_terminated_length": 153.0, "entropy": 0.330673573538661, "epoch": 0.1302968229103908, "frac_reward_zero_std": 0.0, "grad_norm": 1.8808863162994385, "learning_rate": 1.7272727272727275e-07, "loss": -0.0034, "num_tokens": 7389632.0, "reward": 0.998309850692749, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998309850692749, "reward_meter_std": 0.0023174018133431673, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.002317406702786684, "reward_total_composite_mean": 0.998309850692749, "reward_total_composite_std": 0.0023174018133431673, "reward_total_mean": 0.998309850692749, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998309850692749, "rewards/meter/std": 0.0023174018133431673, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998309850692749, "rewards/total_composite/std": 0.0023174018133431673, "sampling/importance_sampling_ratio/max": 1.5186190605163574, "sampling/importance_sampling_ratio/mean": 1.0025235414505005, "sampling/importance_sampling_ratio/min": 0.2143809050321579, "sampling/sampling_logp_difference/max": 1.5400009155273438, "sampling/sampling_logp_difference/mean": 0.039463937282562256, "step": 3244 }, { "clip_ratio/high_max": 0.01926107343751937, "clip_ratio/high_mean": 0.01926107343751937, "clip_ratio/low_mean": 0.006537176435813308, "clip_ratio/low_min": 0.006537176435813308, "clip_ratio/region_mean": 0.02579824987333268, "completions/clipped_ratio": 0.0, "completions/max_length": 140.0, "completions/max_terminated_length": 140.0, "completions/mean_length": 135.375, "completions/mean_terminated_length": 135.375, "completions/min_length": 133.0, "completions/min_terminated_length": 133.0, "entropy": 0.206793999299407, "epoch": 0.13033698839217575, "frac_reward_zero_std": 0.0, "grad_norm": 2.231248378753662, "learning_rate": 1.6969696969696974e-07, "loss": -0.0017, "num_tokens": 7392187.0, "reward": 0.9632641077041626, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989526271820068, "reward_meter_std": 0.0006340565159916878, "reward_repeat_penalty_mean": 0.9642857313156128, "reward_repeat_penalty_std": 0.06613000482320786, "reward_std": 0.0658845454454422, "reward_total_composite_mean": 0.9632641077041626, "reward_total_composite_std": 0.0658845379948616, "reward_total_mean": 0.9632641077041626, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989526271820068, "rewards/meter/std": 0.0006340565159916878, "rewards/repeat_penalty/mean": 0.9642857313156128, "rewards/repeat_penalty/std": 0.06613000482320786, "rewards/total_composite/mean": 0.9632641077041626, "rewards/total_composite/std": 0.0658845379948616, "sampling/importance_sampling_ratio/max": 1.8710554838180542, "sampling/importance_sampling_ratio/mean": 1.0051708221435547, "sampling/importance_sampling_ratio/min": 0.4276849031448364, "sampling/sampling_logp_difference/max": 0.8493685722351074, "sampling/sampling_logp_difference/mean": 0.02810845896601677, "step": 3245 }, { "clip_ratio/high_max": 0.01162644021678716, "clip_ratio/high_mean": 0.01162644021678716, "clip_ratio/low_mean": 0.010340243112295866, "clip_ratio/low_min": 0.010340243112295866, "clip_ratio/region_mean": 0.021966683329083025, "completions/clipped_ratio": 0.0, "completions/max_length": 288.0, "completions/max_terminated_length": 288.0, "completions/mean_length": 270.75, "completions/mean_terminated_length": 270.75, "completions/min_length": 259.0, "completions/min_terminated_length": 259.0, "entropy": 0.38703763484954834, "epoch": 0.1303771538739607, "frac_reward_zero_std": 0.0, "grad_norm": 1.8668049573898315, "learning_rate": 1.6666666666666668e-07, "loss": 0.0198, "num_tokens": 7395817.0, "reward": 0.8224383592605591, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9375, "reward_count_adherence_std": 0.06681530922651291, "reward_meter_mean": 0.9602240324020386, "reward_meter_std": 0.09070871770381927, "reward_repeat_penalty_mean": 0.915900707244873, "reward_repeat_penalty_std": 0.0559099018573761, "reward_std": 0.09149649739265442, "reward_total_composite_mean": 0.8224383592605591, "reward_total_composite_std": 0.09149650484323502, "reward_total_mean": 0.8224383592605591, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9375, "rewards/count_adherence/std": 0.06681530922651291, "rewards/meter/mean": 0.9602240324020386, "rewards/meter/std": 0.09070871770381927, "rewards/repeat_penalty/mean": 0.915900707244873, "rewards/repeat_penalty/std": 0.0559099018573761, "rewards/total_composite/mean": 0.8224383592605591, "rewards/total_composite/std": 0.09149650484323502, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.01006281375885, "sampling/importance_sampling_ratio/min": 0.13546983897686005, "sampling/sampling_logp_difference/max": 1.9990062713623047, "sampling/sampling_logp_difference/mean": 0.03546025604009628, "step": 3246 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.004098360426723957, "clip_ratio/low_min": 0.004098360426723957, "clip_ratio/region_mean": 0.004098360426723957, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.012363885063678026, "epoch": 0.13041731935574566, "frac_reward_zero_std": 0.0, "grad_norm": 0.025763750076293945, "learning_rate": 1.6363636363636367e-07, "loss": 0.0001, "num_tokens": 7397609.0, "reward": 0.9973357319831848, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973357319831848, "reward_meter_std": 8.977292964118533e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.97126938070869e-06, "reward_total_composite_mean": 0.9973357319831848, "reward_total_composite_std": 8.977292964118533e-06, "reward_total_mean": 0.9973357319831848, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973357319831848, "rewards/meter/std": 8.977292964118533e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973357319831848, "rewards/total_composite/std": 8.977292964118533e-06, "sampling/importance_sampling_ratio/max": 1.6409376859664917, "sampling/importance_sampling_ratio/mean": 1.0012048482894897, "sampling/importance_sampling_ratio/min": 0.6253011226654053, "sampling/sampling_logp_difference/max": 0.4952678680419922, "sampling/sampling_logp_difference/mean": 0.0030104033648967743, "step": 3247 }, { "clip_ratio/high_max": 0.023729265900328755, "clip_ratio/high_mean": 0.023729265900328755, "clip_ratio/low_mean": 0.0085039883852005, "clip_ratio/low_min": 0.0085039883852005, "clip_ratio/region_mean": 0.032233254285529256, "completions/clipped_ratio": 0.0, "completions/max_length": 173.0, "completions/max_terminated_length": 173.0, "completions/mean_length": 166.75, "completions/mean_terminated_length": 166.75, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.3529275842010975, "epoch": 0.1304574848375306, "frac_reward_zero_std": 0.0, "grad_norm": 4.575743675231934, "learning_rate": 1.606060606060606e-07, "loss": -0.0118, "num_tokens": 7400383.0, "reward": 0.9712226390838623, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9989683628082275, "reward_meter_std": 0.00019462761702015996, "reward_repeat_penalty_mean": 0.9722222089767456, "reward_repeat_penalty_std": 0.05143444985151291, "reward_std": 0.051451049745082855, "reward_total_composite_mean": 0.9712226390838623, "reward_total_composite_std": 0.051451049745082855, "reward_total_mean": 0.9712226390838623, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9989683628082275, "rewards/meter/std": 0.00019462761702015996, "rewards/repeat_penalty/mean": 0.9722222089767456, "rewards/repeat_penalty/std": 0.05143444985151291, "rewards/total_composite/mean": 0.9712226390838623, "rewards/total_composite/std": 0.051451049745082855, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0083783864974976, "sampling/importance_sampling_ratio/min": 0.21651406586170197, "sampling/sampling_logp_difference/max": 1.530099868774414, "sampling/sampling_logp_difference/mean": 0.040822967886924744, "step": 3248 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 1.0, "completions/max_length": 512.0, "completions/max_terminated_length": 0.0, "completions/mean_length": 512.0, "completions/mean_terminated_length": 0.0, "completions/min_length": 512.0, "completions/min_terminated_length": 0.0, "entropy": 0.0, "epoch": 0.13049765031931557, "frac_reward_zero_std": 0.0, "grad_norm": 0.0, "learning_rate": 1.575757575757576e-07, "loss": 0.0, "num_tokens": 7402303.0, "reward": 0.7142915725708008, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.7291666269302368, "reward_count_adherence_std": 0.01964186504483223, "reward_meter_mean": 0.998656690120697, "reward_meter_std": 0.0011350500863045454, "reward_repeat_penalty_mean": 0.9807692170143127, "reward_repeat_penalty_std": 0.02907419577240944, "reward_std": 0.03164464607834816, "reward_total_composite_mean": 0.7142915725708008, "reward_total_composite_std": 0.03164464980363846, "reward_total_mean": 0.7142915725708008, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.7291666269302368, "rewards/count_adherence/std": 0.01964186504483223, "rewards/meter/mean": 0.998656690120697, "rewards/meter/std": 0.0011350500863045454, "rewards/repeat_penalty/mean": 0.9807692170143127, "rewards/repeat_penalty/std": 0.02907419577240944, "rewards/total_composite/mean": 0.7142915725708008, "rewards/total_composite/std": 0.03164464980363846, "sampling/importance_sampling_ratio/max": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/sampling_logp_difference/mean": 0.0, "step": 3249 }, { "clip_ratio/high_max": 0.00882849539630115, "clip_ratio/high_mean": 0.00882849539630115, "clip_ratio/low_mean": 0.010569972451776266, "clip_ratio/low_min": 0.010569972451776266, "clip_ratio/region_mean": 0.019398467848077416, "completions/clipped_ratio": 0.0, "completions/max_length": 143.0, "completions/max_terminated_length": 143.0, "completions/mean_length": 141.875, "completions/mean_terminated_length": 141.875, "completions/min_length": 140.0, "completions/min_terminated_length": 140.0, "entropy": 0.20848924107849598, "epoch": 0.13053781580110052, "frac_reward_zero_std": 0.0, "grad_norm": 1.1469806432724, "learning_rate": 1.5454545454545456e-07, "loss": -0.0012, "num_tokens": 7404902.0, "reward": 0.9991378784179688, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9991378784179688, "reward_meter_std": 0.00011923787678824738, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00011922277917619795, "reward_total_composite_mean": 0.9991378784179688, "reward_total_composite_std": 0.00011923787678824738, "reward_total_mean": 0.9991378784179688, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9991378784179688, "rewards/meter/std": 0.00011923787678824738, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9991378784179688, "rewards/total_composite/std": 0.00011923787678824738, "sampling/importance_sampling_ratio/max": 1.4649658203125, "sampling/importance_sampling_ratio/mean": 1.0037734508514404, "sampling/importance_sampling_ratio/min": 0.285825252532959, "sampling/sampling_logp_difference/max": 1.2523746490478516, "sampling/sampling_logp_difference/mean": 0.020554685965180397, "step": 3250 }, { "epoch": 0.13053781580110052, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.028846153846153848, "eval_completions/max_length": 421.38461538461536, "eval_completions/max_terminated_length": 398.7692307692308, "eval_completions/mean_length": 214.33653846153845, "eval_completions/mean_terminated_length": 204.00686880258414, "eval_completions/min_length": 60.76923076923077, "eval_completions/min_terminated_length": 60.76923076923077, "eval_entropy": 0.38327448299297917, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 7404902.0, "eval_reward": 0.7313130910579975, "eval_reward_arabic_clean_mean": 0.9807692307692307, "eval_reward_arabic_clean_std": 0.05439282839114849, "eval_reward_count_adherence_mean": 0.95811671935595, "eval_reward_count_adherence_std": 0.06441009388520168, "eval_reward_meter_mean": 0.7977132155345037, "eval_reward_meter_std": 0.3206370784542881, "eval_reward_repeat_penalty_mean": 0.9612986537126395, "eval_reward_repeat_penalty_std": 0.06227556343835134, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7313130910579975, "eval_reward_total_composite_std": 0.33714570047763676, "eval_reward_total_mean": 0.7313130910579975, "eval_rewards/arabic_clean/mean": 0.9807692307692307, "eval_rewards/arabic_clean/std": 0.05439282839114849, "eval_rewards/count_adherence/mean": 0.95811671935595, "eval_rewards/count_adherence/std": 0.06441009388520168, "eval_rewards/meter/mean": 0.7977132155345037, "eval_rewards/meter/std": 0.3206370784542881, "eval_rewards/repeat_penalty/mean": 0.9612986537126395, "eval_rewards/repeat_penalty/std": 0.06227556343835134, "eval_rewards/total_composite/mean": 0.7313130910579975, "eval_rewards/total_composite/std": 0.33714570047763676, "eval_runtime": 78.0663, "eval_samples_per_second": 1.332, "eval_sampling/importance_sampling_ratio/max": 1.4786042525218084, "eval_sampling/importance_sampling_ratio/mean": 1.0085856593572176, "eval_sampling/importance_sampling_ratio/min": 0.30588124233942765, "eval_sampling/sampling_logp_difference/max": 1.2143818781926081, "eval_sampling/sampling_logp_difference/mean": 0.034000108448358685, "eval_steps_per_second": 0.167, "step": 3250 }, { "clip_ratio/high_max": 0.009146341122686863, "clip_ratio/high_mean": 0.009146341122686863, "clip_ratio/low_mean": 0.006253908621147275, "clip_ratio/low_min": 0.006253908621147275, "clip_ratio/region_mean": 0.015400249743834138, "completions/clipped_ratio": 0.0, "completions/max_length": 41.0, "completions/max_terminated_length": 41.0, "completions/mean_length": 40.75, "completions/mean_terminated_length": 40.75, "completions/min_length": 39.0, "completions/min_terminated_length": 39.0, "entropy": 0.12377766706049442, "epoch": 0.13057798128288547, "frac_reward_zero_std": 0.0, "grad_norm": 1.9768705368041992, "learning_rate": 1.5151515151515152e-07, "loss": 0.003, "num_tokens": 7406604.0, "reward": 0.998726487159729, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.998726487159729, "reward_meter_std": 0.0001513757451903075, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001513654860900715, "reward_total_composite_mean": 0.998726487159729, "reward_total_composite_std": 0.0001513757451903075, "reward_total_mean": 0.998726487159729, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.998726487159729, "rewards/meter/std": 0.0001513757451903075, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.998726487159729, "rewards/total_composite/std": 0.0001513757451903075, "sampling/importance_sampling_ratio/max": 1.5404958724975586, "sampling/importance_sampling_ratio/mean": 0.9942156076431274, "sampling/importance_sampling_ratio/min": 0.275457501411438, "sampling/sampling_logp_difference/max": 1.2893218994140625, "sampling/sampling_logp_difference/mean": 0.020130213350057602, "step": 3251 }, { "clip_ratio/high_max": 0.012048649485222995, "clip_ratio/high_mean": 0.012048649485222995, "clip_ratio/low_mean": 0.0009920635493472219, "clip_ratio/low_min": 0.0009920635493472219, "clip_ratio/region_mean": 0.013040713034570217, "completions/clipped_ratio": 0.0, "completions/max_length": 126.0, "completions/max_terminated_length": 126.0, "completions/mean_length": 124.75, "completions/mean_terminated_length": 124.75, "completions/min_length": 123.0, "completions/min_terminated_length": 123.0, "entropy": 0.12918496038764715, "epoch": 0.13061814676467043, "frac_reward_zero_std": 0.0, "grad_norm": 3.5564746856689453, "learning_rate": 1.484848484848485e-07, "loss": 0.0033, "num_tokens": 7409050.0, "reward": 0.9973940253257751, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973940253257751, "reward_meter_std": 0.0014456679346039891, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.001445673406124115, "reward_total_composite_mean": 0.9973940253257751, "reward_total_composite_std": 0.0014456679346039891, "reward_total_mean": 0.9973940253257751, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973940253257751, "rewards/meter/std": 0.0014456679346039891, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973940253257751, "rewards/total_composite/std": 0.0014456679346039891, "sampling/importance_sampling_ratio/max": 1.582746148109436, "sampling/importance_sampling_ratio/mean": 1.0002362728118896, "sampling/importance_sampling_ratio/min": 0.22844921052455902, "sampling/sampling_logp_difference/max": 1.4764413833618164, "sampling/sampling_logp_difference/mean": 0.018116910010576248, "step": 3252 }, { "clip_ratio/high_max": 0.017753356718458235, "clip_ratio/high_mean": 0.017753356718458235, "clip_ratio/low_mean": 0.002419354859739542, "clip_ratio/low_min": 0.002419354859739542, "clip_ratio/region_mean": 0.020172711578197777, "completions/clipped_ratio": 0.0, "completions/max_length": 156.0, "completions/max_terminated_length": 156.0, "completions/mean_length": 154.375, "completions/mean_terminated_length": 154.375, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "entropy": 0.2324562668800354, "epoch": 0.13065831224645538, "frac_reward_zero_std": 0.0, "grad_norm": 1.9089981317520142, "learning_rate": 1.4545454545454548e-07, "loss": 0.008, "num_tokens": 7411813.0, "reward": 0.983467161655426, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973270297050476, "reward_meter_std": 0.0007912821019999683, "reward_repeat_penalty_mean": 0.9861111044883728, "reward_repeat_penalty_std": 0.03928370773792267, "reward_std": 0.03897340968251228, "reward_total_composite_mean": 0.983467161655426, "reward_total_composite_std": 0.038973402231931686, "reward_total_mean": 0.983467161655426, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973270297050476, "rewards/meter/std": 0.0007912821019999683, "rewards/repeat_penalty/mean": 0.9861111044883728, "rewards/repeat_penalty/std": 0.03928370773792267, "rewards/total_composite/mean": 0.983467161655426, "rewards/total_composite/std": 0.038973402231931686, "sampling/importance_sampling_ratio/max": 1.7957075834274292, "sampling/importance_sampling_ratio/mean": 1.0039182901382446, "sampling/importance_sampling_ratio/min": 0.24512901902198792, "sampling/sampling_logp_difference/max": 1.405970573425293, "sampling/sampling_logp_difference/mean": 0.029868151992559433, "step": 3253 }, { "clip_ratio/high_max": 0.025140726240351796, "clip_ratio/high_mean": 0.025140726240351796, "clip_ratio/low_mean": 0.007630055071786046, "clip_ratio/low_min": 0.007630055071786046, "clip_ratio/region_mean": 0.03277078131213784, "completions/clipped_ratio": 0.0, "completions/max_length": 241.0, "completions/max_terminated_length": 241.0, "completions/mean_length": 232.375, "completions/mean_terminated_length": 232.375, "completions/min_length": 227.0, "completions/min_terminated_length": 227.0, "entropy": 0.34237322956323624, "epoch": 0.13069847772824034, "frac_reward_zero_std": 0.0, "grad_norm": 2.1981167793273926, "learning_rate": 1.4242424242424244e-07, "loss": -0.0002, "num_tokens": 7415280.0, "reward": 0.978593111038208, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9978033304214478, "reward_meter_std": 0.0022864267230033875, "reward_repeat_penalty_mean": 0.9807692170143127, "reward_repeat_penalty_std": 0.03560846298933029, "reward_std": 0.03494199737906456, "reward_total_composite_mean": 0.978593111038208, "reward_total_composite_std": 0.03494200482964516, "reward_total_mean": 0.978593111038208, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9978033304214478, "rewards/meter/std": 0.0022864267230033875, "rewards/repeat_penalty/mean": 0.9807692170143127, "rewards/repeat_penalty/std": 0.03560846298933029, "rewards/total_composite/mean": 0.978593111038208, "rewards/total_composite/std": 0.03494200482964516, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0061860084533691, "sampling/importance_sampling_ratio/min": 0.2894412875175476, "sampling/sampling_logp_difference/max": 1.2398028373718262, "sampling/sampling_logp_difference/mean": 0.042390793561935425, "step": 3254 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.000789376012107823, "epoch": 0.1307386432100253, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.393939393939394e-07, "loss": 0.0, "num_tokens": 7417000.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0015891790390015, "sampling/importance_sampling_ratio/mean": 1.0000584125518799, "sampling/importance_sampling_ratio/min": 0.9946505427360535, "sampling/sampling_logp_difference/max": 0.005363806616514921, "sampling/sampling_logp_difference/mean": 8.257458102889359e-05, "step": 3255 }, { "clip_ratio/high_max": 0.01858959416858852, "clip_ratio/high_mean": 0.01858959416858852, "clip_ratio/low_mean": 0.0012376237427815795, "clip_ratio/low_min": 0.0012376237427815795, "clip_ratio/region_mean": 0.0198272179113701, "completions/clipped_ratio": 0.0, "completions/max_length": 102.0, "completions/max_terminated_length": 102.0, "completions/mean_length": 100.875, "completions/mean_terminated_length": 100.875, "completions/min_length": 100.0, "completions/min_terminated_length": 100.0, "entropy": 0.13637553714215755, "epoch": 0.13077880869181027, "frac_reward_zero_std": 0.0, "grad_norm": 1.5352351665496826, "learning_rate": 1.3636363636363637e-07, "loss": 0.0005, "num_tokens": 7419231.0, "reward": 0.9743703603744507, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993561506271362, "reward_meter_std": 0.00013285702152643353, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.07064089179039001, "reward_total_composite_mean": 0.9743703603744507, "reward_total_composite_std": 0.07064089924097061, "reward_total_mean": 0.9743703603744507, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993561506271362, "rewards/meter/std": 0.00013285702152643353, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9743703603744507, "rewards/total_composite/std": 0.07064089924097061, "sampling/importance_sampling_ratio/max": 1.6758545637130737, "sampling/importance_sampling_ratio/mean": 1.0004955530166626, "sampling/importance_sampling_ratio/min": 0.2817700207233429, "sampling/sampling_logp_difference/max": 1.2666641473770142, "sampling/sampling_logp_difference/mean": 0.01845725066959858, "step": 3256 }, { "clip_ratio/high_max": 0.008754611015319824, "clip_ratio/high_mean": 0.008754611015319824, "clip_ratio/low_mean": 0.007067404338158667, "clip_ratio/low_min": 0.007067404338158667, "clip_ratio/region_mean": 0.01582201535347849, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.0, "completions/mean_terminated_length": 71.0, "completions/min_length": 70.0, "completions/min_terminated_length": 70.0, "entropy": 0.0983225479722023, "epoch": 0.13081897417359523, "frac_reward_zero_std": 0.0, "grad_norm": 1.7167513370513916, "learning_rate": 1.3333333333333336e-07, "loss": 0.0003, "num_tokens": 7421119.0, "reward": 0.9994316101074219, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994316101074219, "reward_meter_std": 0.0001929309801198542, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00019294493540655822, "reward_total_composite_mean": 0.9994316101074219, "reward_total_composite_std": 0.0001929309801198542, "reward_total_mean": 0.9994316101074219, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994316101074219, "rewards/meter/std": 0.0001929309801198542, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994316101074219, "rewards/total_composite/std": 0.0001929309801198542, "sampling/importance_sampling_ratio/max": 1.2882606983184814, "sampling/importance_sampling_ratio/mean": 0.9987840056419373, "sampling/importance_sampling_ratio/min": 0.31890156865119934, "sampling/sampling_logp_difference/max": 1.1428728103637695, "sampling/sampling_logp_difference/mean": 0.014849050901830196, "step": 3257 }, { "clip_ratio/high_max": 0.01056338008493185, "clip_ratio/high_mean": 0.01056338008493185, "clip_ratio/low_mean": 0.0017605633474886417, "clip_ratio/low_min": 0.0017605633474886417, "clip_ratio/region_mean": 0.012323943432420492, "completions/clipped_ratio": 0.0, "completions/max_length": 71.0, "completions/max_terminated_length": 71.0, "completions/mean_length": 71.0, "completions/mean_terminated_length": 71.0, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.0794384153559804, "epoch": 0.13085913965538018, "frac_reward_zero_std": 0.0, "grad_norm": 0.16614317893981934, "learning_rate": 1.3030303030303033e-07, "loss": 0.0004, "num_tokens": 7422983.0, "reward": 0.9994429349899292, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994429349899292, "reward_meter_std": 1.4887036741129123e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.4891715181875043e-05, "reward_total_composite_mean": 0.9994429349899292, "reward_total_composite_std": 1.4887036741129123e-05, "reward_total_mean": 0.9994429349899292, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994429349899292, "rewards/meter/std": 1.4887036741129123e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994429349899292, "rewards/total_composite/std": 1.4887036741129123e-05, "sampling/importance_sampling_ratio/max": 1.2386420965194702, "sampling/importance_sampling_ratio/mean": 0.9997973442077637, "sampling/importance_sampling_ratio/min": 0.2568560242652893, "sampling/sampling_logp_difference/max": 1.3592395782470703, "sampling/sampling_logp_difference/mean": 0.013734190724790096, "step": 3258 }, { "clip_ratio/high_max": 0.018950162804685533, "clip_ratio/high_mean": 0.018950162804685533, "clip_ratio/low_mean": 0.01923532225191593, "clip_ratio/low_min": 0.01923532225191593, "clip_ratio/region_mean": 0.038185485056601465, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 65.75, "completions/mean_terminated_length": 65.75, "completions/min_length": 63.0, "completions/min_terminated_length": 63.0, "entropy": 0.1472517903894186, "epoch": 0.13089930513716513, "frac_reward_zero_std": 0.0, "grad_norm": 6.062583923339844, "learning_rate": 1.272727272727273e-07, "loss": -0.0021, "num_tokens": 7424765.0, "reward": 0.9497079849243164, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9497079849243164, "reward_meter_std": 0.005244073923677206, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00524406973272562, "reward_total_composite_mean": 0.9497079849243164, "reward_total_composite_std": 0.005244073923677206, "reward_total_mean": 0.9497079849243164, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9497079849243164, "rewards/meter/std": 0.005244073923677206, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9497079849243164, "rewards/total_composite/std": 0.005244073923677206, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0019686222076416, "sampling/importance_sampling_ratio/min": 0.3562839925289154, "sampling/sampling_logp_difference/max": 1.032027244567871, "sampling/sampling_logp_difference/mean": 0.030090492218732834, "step": 3259 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 32.0, "completions/max_terminated_length": 32.0, "completions/mean_length": 32.0, "completions/mean_terminated_length": 32.0, "completions/min_length": 32.0, "completions/min_terminated_length": 32.0, "entropy": 0.0003866630977427121, "epoch": 0.1309394706189501, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.2424242424242426e-07, "loss": 0.0, "num_tokens": 7426277.0, "reward": 0.9992982149124146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992982149124146, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9992982149124146, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9992982149124146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992982149124146, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992982149124146, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0006037950515747, "sampling/importance_sampling_ratio/mean": 1.0000357627868652, "sampling/importance_sampling_ratio/min": 0.9986362457275391, "sampling/sampling_logp_difference/max": 0.001364716561511159, "sampling/sampling_logp_difference/mean": 5.620270167128183e-05, "step": 3260 }, { "clip_ratio/high_max": 0.012933629681356251, "clip_ratio/high_mean": 0.012933629681356251, "clip_ratio/low_mean": 0.007126289419829845, "clip_ratio/low_min": 0.007126289419829845, "clip_ratio/region_mean": 0.020059919101186097, "completions/clipped_ratio": 0.0, "completions/max_length": 74.0, "completions/max_terminated_length": 74.0, "completions/mean_length": 69.625, "completions/mean_terminated_length": 69.625, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.1911148466169834, "epoch": 0.13097963610073504, "frac_reward_zero_std": 0.0, "grad_norm": 3.223155975341797, "learning_rate": 1.2121212121212122e-07, "loss": 0.0044, "num_tokens": 7428154.0, "reward": 0.9926314949989319, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9926314949989319, "reward_meter_std": 0.006077465135604143, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.006077463272958994, "reward_total_composite_mean": 0.9926314949989319, "reward_total_composite_std": 0.006077465135604143, "reward_total_mean": 0.9926314949989319, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9926314949989319, "rewards/meter/std": 0.006077465135604143, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926314949989319, "rewards/total_composite/std": 0.006077465135604143, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0040087699890137, "sampling/importance_sampling_ratio/min": 0.31960853934288025, "sampling/sampling_logp_difference/max": 1.1406583786010742, "sampling/sampling_logp_difference/mean": 0.029856102541089058, "step": 3261 }, { "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/low_mean": 0.0034966744715347886, "clip_ratio/low_min": 0.0034966744715347886, "clip_ratio/region_mean": 0.00525723781902343, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.125, "completions/mean_terminated_length": 71.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.07193375751376152, "epoch": 0.13101980158252, "frac_reward_zero_std": 0.0, "grad_norm": 0.49830862879753113, "learning_rate": 1.1818181818181818e-07, "loss": 0.0013, "num_tokens": 7429891.0, "reward": 0.9994339346885681, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994339346885681, "reward_meter_std": 2.7899814085685648e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 2.791790757328272e-05, "reward_total_composite_mean": 0.9994339346885681, "reward_total_composite_std": 2.7899814085685648e-05, "reward_total_mean": 0.9994339346885681, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994339346885681, "rewards/meter/std": 2.7899814085685648e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994339346885681, "rewards/total_composite/std": 2.7899814085685648e-05, "sampling/importance_sampling_ratio/max": 1.19877290725708, "sampling/importance_sampling_ratio/mean": 1.0001577138900757, "sampling/importance_sampling_ratio/min": 0.19248820841312408, "sampling/sampling_logp_difference/max": 1.6477203369140625, "sampling/sampling_logp_difference/mean": 0.011753112077713013, "step": 3262 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.012014997191727161, "epoch": 0.13105996706430495, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.1515151515151516e-07, "loss": 0.0, "num_tokens": 7431611.0, "reward": 0.9973388910293579, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973388910293579, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9973388910293579, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9973388910293579, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973388910293579, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973388910293579, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0381407737731934, "sampling/importance_sampling_ratio/mean": 1.0003302097320557, "sampling/importance_sampling_ratio/min": 0.9129676818847656, "sampling/sampling_logp_difference/max": 0.09105479717254639, "sampling/sampling_logp_difference/mean": 0.0008966223103925586, "step": 3263 }, { "clip_ratio/high_max": 0.003839680110104382, "clip_ratio/high_mean": 0.003839680110104382, "clip_ratio/low_mean": 0.0038265305338427424, "clip_ratio/low_min": 0.0038265305338427424, "clip_ratio/region_mean": 0.0076662106439471245, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 97.875, "completions/mean_terminated_length": 97.875, "completions/min_length": 97.0, "completions/min_terminated_length": 97.0, "entropy": 0.03584319236688316, "epoch": 0.1311001325460899, "frac_reward_zero_std": 0.0, "grad_norm": 0.04755792394280434, "learning_rate": 1.1212121212121213e-07, "loss": 0.0001, "num_tokens": 7433682.0, "reward": 0.999409019947052, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999409019947052, "reward_meter_std": 5.810459242638899e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 5.812751624034718e-06, "reward_total_composite_mean": 0.999409019947052, "reward_total_composite_std": 5.810459242638899e-06, "reward_total_mean": 0.999409019947052, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999409019947052, "rewards/meter/std": 5.810459242638899e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999409019947052, "rewards/total_composite/std": 5.810459242638899e-06, "sampling/importance_sampling_ratio/max": 1.4332921504974365, "sampling/importance_sampling_ratio/mean": 1.0018383264541626, "sampling/importance_sampling_ratio/min": 0.7912850379943848, "sampling/sampling_logp_difference/max": 0.3599740266799927, "sampling/sampling_logp_difference/mean": 0.003859634278342128, "step": 3264 }, { "clip_ratio/high_max": 0.015736370929516852, "clip_ratio/high_mean": 0.015736370929516852, "clip_ratio/low_mean": 0.011210258584469557, "clip_ratio/low_min": 0.011210258584469557, "clip_ratio/region_mean": 0.02694662951398641, "completions/clipped_ratio": 0.0, "completions/max_length": 169.0, "completions/max_terminated_length": 169.0, "completions/mean_length": 166.875, "completions/mean_terminated_length": 166.875, "completions/min_length": 162.0, "completions/min_terminated_length": 162.0, "entropy": 0.31825447641313076, "epoch": 0.13114029802787486, "frac_reward_zero_std": 0.0, "grad_norm": 2.038606643676758, "learning_rate": 1.090909090909091e-07, "loss": 0.0039, "num_tokens": 7436489.0, "reward": 0.999045729637146, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999045729637146, "reward_meter_std": 0.00020887328719254583, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00020887046412099153, "reward_total_composite_mean": 0.999045729637146, "reward_total_composite_std": 0.00020887328719254583, "reward_total_mean": 0.999045729637146, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999045729637146, "rewards/meter/std": 0.00020887328719254583, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999045729637146, "rewards/total_composite/std": 0.00020887328719254583, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0036981105804443, "sampling/importance_sampling_ratio/min": 0.21604275703430176, "sampling/sampling_logp_difference/max": 1.5322790145874023, "sampling/sampling_logp_difference/mean": 0.03505001217126846, "step": 3265 }, { "clip_ratio/high_max": 0.018389825709164143, "clip_ratio/high_mean": 0.018389825709164143, "clip_ratio/low_mean": 0.003257433301769197, "clip_ratio/low_min": 0.003257433301769197, "clip_ratio/region_mean": 0.02164725901093334, "completions/clipped_ratio": 0.0, "completions/max_length": 159.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 155.375, "completions/mean_terminated_length": 155.375, "completions/min_length": 152.0, "completions/min_terminated_length": 152.0, "entropy": 0.30485600233078003, "epoch": 0.1311804635096598, "frac_reward_zero_std": 0.0, "grad_norm": 1.5206332206726074, "learning_rate": 1.0606060606060608e-07, "loss": -0.003, "num_tokens": 7439156.0, "reward": 0.9990906119346619, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990906119346619, "reward_meter_std": 0.0002633438853081316, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002633513940963894, "reward_total_composite_mean": 0.9990906119346619, "reward_total_composite_std": 0.0002633438853081316, "reward_total_mean": 0.9990906119346619, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990906119346619, "rewards/meter/std": 0.0002633438853081316, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990906119346619, "rewards/total_composite/std": 0.0002633438853081316, "sampling/importance_sampling_ratio/max": 1.7514793872833252, "sampling/importance_sampling_ratio/mean": 1.009647250175476, "sampling/importance_sampling_ratio/min": 0.27048155665397644, "sampling/sampling_logp_difference/max": 1.307551383972168, "sampling/sampling_logp_difference/mean": 0.031446803361177444, "step": 3266 }, { "clip_ratio/high_max": 0.011238692968618125, "clip_ratio/high_mean": 0.011238692968618125, "clip_ratio/low_mean": 0.004726932500489056, "clip_ratio/low_min": 0.004726932500489056, "clip_ratio/region_mean": 0.01596562546910718, "completions/clipped_ratio": 0.0, "completions/max_length": 159.0, "completions/max_terminated_length": 159.0, "completions/mean_length": 156.5, "completions/mean_terminated_length": 156.5, "completions/min_length": 155.0, "completions/min_terminated_length": 155.0, "entropy": 0.3256886452436447, "epoch": 0.13122062899144477, "frac_reward_zero_std": 0.0, "grad_norm": 1.932774305343628, "learning_rate": 1.0303030303030304e-07, "loss": 0.0021, "num_tokens": 7441984.0, "reward": 0.9990622997283936, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990622997283936, "reward_meter_std": 0.00025987729895859957, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0002598702849354595, "reward_total_composite_mean": 0.9990622997283936, "reward_total_composite_std": 0.00025987729895859957, "reward_total_mean": 0.9990622997283936, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990622997283936, "rewards/meter/std": 0.00025987729895859957, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990622997283936, "rewards/total_composite/std": 0.00025987729895859957, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0105011463165283, "sampling/importance_sampling_ratio/min": 0.29758986830711365, "sampling/sampling_logp_difference/max": 1.2120389938354492, "sampling/sampling_logp_difference/mean": 0.035560522228479385, "step": 3267 }, { "clip_ratio/high_max": 0.005740093358326703, "clip_ratio/high_mean": 0.005740093358326703, "clip_ratio/low_mean": 0.0019083969527855515, "clip_ratio/low_min": 0.0019083969527855515, "clip_ratio/region_mean": 0.007648490311112255, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 131.25, "completions/mean_terminated_length": 131.25, "completions/min_length": 130.0, "completions/min_terminated_length": 130.0, "entropy": 0.08114278595894575, "epoch": 0.13126079447322972, "frac_reward_zero_std": 0.0, "grad_norm": 0.6131901741027832, "learning_rate": 1.0000000000000001e-07, "loss": -0.0011, "num_tokens": 7444602.0, "reward": 0.9992755651473999, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992755651473999, "reward_meter_std": 0.00033690276904962957, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0003369078040122986, "reward_total_composite_mean": 0.9992755651473999, "reward_total_composite_std": 0.00033690276904962957, "reward_total_mean": 0.9992755651473999, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992755651473999, "rewards/meter/std": 0.00033690276904962957, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992755651473999, "rewards/total_composite/std": 0.00033690276904962957, "sampling/importance_sampling_ratio/max": 1.4053065776824951, "sampling/importance_sampling_ratio/mean": 0.9999168515205383, "sampling/importance_sampling_ratio/min": 0.1141020655632019, "sampling/sampling_logp_difference/max": 2.1706619262695312, "sampling/sampling_logp_difference/mean": 0.011317918077111244, "step": 3268 }, { "clip_ratio/high_max": 0.00815217406488955, "clip_ratio/high_mean": 0.00815217406488955, "clip_ratio/low_mean": 0.00815217406488955, "clip_ratio/low_min": 0.00815217406488955, "clip_ratio/region_mean": 0.0163043481297791, "completions/clipped_ratio": 0.0, "completions/max_length": 46.0, "completions/max_terminated_length": 46.0, "completions/mean_length": 46.0, "completions/mean_terminated_length": 46.0, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.08023244515061378, "epoch": 0.13130095995501467, "frac_reward_zero_std": 0.0, "grad_norm": 3.9373042583465576, "learning_rate": 9.696969696969697e-08, "loss": 0.0089, "num_tokens": 7446314.0, "reward": 0.9440339803695679, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9440339803695679, "reward_meter_std": 0.0030637700110673904, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0030637700110673904, "reward_total_composite_mean": 0.9440339803695679, "reward_total_composite_std": 0.0030637700110673904, "reward_total_mean": 0.9440339803695679, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9440339803695679, "rewards/meter/std": 0.0030637700110673904, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9440339803695679, "rewards/total_composite/std": 0.0030637700110673904, "sampling/importance_sampling_ratio/max": 1.5834907293319702, "sampling/importance_sampling_ratio/mean": 1.0039082765579224, "sampling/importance_sampling_ratio/min": 0.20156030356884003, "sampling/sampling_logp_difference/max": 1.6016666889190674, "sampling/sampling_logp_difference/mean": 0.020654289051890373, "step": 3269 }, { "clip_ratio/high_max": 0.016811800538562238, "clip_ratio/high_mean": 0.016811800538562238, "clip_ratio/low_mean": 0.005398246110416949, "clip_ratio/low_min": 0.005398246110416949, "clip_ratio/region_mean": 0.022210046648979187, "completions/clipped_ratio": 0.0, "completions/max_length": 238.0, "completions/max_terminated_length": 238.0, "completions/mean_length": 232.625, "completions/mean_terminated_length": 232.625, "completions/min_length": 226.0, "completions/min_terminated_length": 226.0, "entropy": 0.4155345596373081, "epoch": 0.13134112543679963, "frac_reward_zero_std": 0.0, "grad_norm": 2.0776824951171875, "learning_rate": 9.393939393939395e-08, "loss": 0.0097, "num_tokens": 7449775.0, "reward": 0.9988599419593811, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9988599419593811, "reward_meter_std": 0.0006400212296284735, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0006400320562534034, "reward_total_composite_mean": 0.9988599419593811, "reward_total_composite_std": 0.0006400212296284735, "reward_total_mean": 0.9988599419593811, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9988599419593811, "rewards/meter/std": 0.0006400212296284735, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9988599419593811, "rewards/total_composite/std": 0.0006400212296284735, "sampling/importance_sampling_ratio/max": 1.781883716583252, "sampling/importance_sampling_ratio/mean": 1.0114996433258057, "sampling/importance_sampling_ratio/min": 0.2545531392097473, "sampling/sampling_logp_difference/max": 1.3682456016540527, "sampling/sampling_logp_difference/mean": 0.04505442455410957, "step": 3270 }, { "clip_ratio/high_max": 0.0093678361736238, "clip_ratio/high_mean": 0.0093678361736238, "clip_ratio/low_mean": 0.004673101240769029, "clip_ratio/low_min": 0.004673101240769029, "clip_ratio/region_mean": 0.014040937414392829, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 106.875, "completions/mean_terminated_length": 106.875, "completions/min_length": 106.0, "completions/min_terminated_length": 106.0, "entropy": 0.16768567822873592, "epoch": 0.13138129091858458, "frac_reward_zero_std": 0.0, "grad_norm": 0.9732382893562317, "learning_rate": 9.090909090909091e-08, "loss": 0.0018, "num_tokens": 7451918.0, "reward": 0.9993022680282593, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993022680282593, "reward_meter_std": 8.71433803695254e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.713654096936807e-05, "reward_total_composite_mean": 0.9993022680282593, "reward_total_composite_std": 8.71433803695254e-05, "reward_total_mean": 0.9993022680282593, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993022680282593, "rewards/meter/std": 8.71433803695254e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993022680282593, "rewards/total_composite/std": 8.71433803695254e-05, "sampling/importance_sampling_ratio/max": 1.4888734817504883, "sampling/importance_sampling_ratio/mean": 1.00331711769104, "sampling/importance_sampling_ratio/min": 0.29764947295188904, "sampling/sampling_logp_difference/max": 1.211838722229004, "sampling/sampling_logp_difference/mean": 0.016620170325040817, "step": 3271 }, { "clip_ratio/high_max": 0.017071759328246117, "clip_ratio/high_mean": 0.017071759328246117, "clip_ratio/low_mean": 0.008297410036902875, "clip_ratio/low_min": 0.008297410036902875, "clip_ratio/region_mean": 0.02536916936514899, "completions/clipped_ratio": 0.0, "completions/max_length": 144.0, "completions/max_terminated_length": 144.0, "completions/mean_length": 136.5, "completions/mean_terminated_length": 136.5, "completions/min_length": 132.0, "completions/min_terminated_length": 132.0, "entropy": 0.310006458312273, "epoch": 0.13142145640036954, "frac_reward_zero_std": 0.0, "grad_norm": 2.9168341159820557, "learning_rate": 8.787878787878788e-08, "loss": -0.0028, "num_tokens": 7454434.0, "reward": 0.9926499128341675, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9926499128341675, "reward_meter_std": 0.0036937205586582422, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0036937135737389326, "reward_total_composite_mean": 0.9926499128341675, "reward_total_composite_std": 0.0036937205586582422, "reward_total_mean": 0.9926499128341675, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9926499128341675, "rewards/meter/std": 0.0036937205586582422, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9926499128341675, "rewards/total_composite/std": 0.0036937205586582422, "sampling/importance_sampling_ratio/max": 1.8989595174789429, "sampling/importance_sampling_ratio/mean": 1.0109081268310547, "sampling/importance_sampling_ratio/min": 0.31607571244239807, "sampling/sampling_logp_difference/max": 1.151773452758789, "sampling/sampling_logp_difference/mean": 0.0317571647465229, "step": 3272 }, { "clip_ratio/high_max": 0.005597014795057476, "clip_ratio/high_mean": 0.005597014795057476, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.005597014795057476, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 67.0, "completions/mean_terminated_length": 67.0, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.04388477047905326, "epoch": 0.1314616218821545, "frac_reward_zero_std": 0.0, "grad_norm": 2.5891594886779785, "learning_rate": 8.484848484848487e-08, "loss": 0.0013, "num_tokens": 7456266.0, "reward": 0.9981439113616943, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981439113616943, "reward_meter_std": 3.332228152430616e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.332816413603723e-05, "reward_total_composite_mean": 0.9981439113616943, "reward_total_composite_std": 3.332228152430616e-05, "reward_total_mean": 0.9981439113616943, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981439113616943, "rewards/meter/std": 3.332228152430616e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981439113616943, "rewards/total_composite/std": 3.332228152430616e-05, "sampling/importance_sampling_ratio/max": 1.3197448253631592, "sampling/importance_sampling_ratio/mean": 0.9988389611244202, "sampling/importance_sampling_ratio/min": 0.4461836516857147, "sampling/sampling_logp_difference/max": 0.8070247173309326, "sampling/sampling_logp_difference/mean": 0.0074036987498402596, "step": 3273 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0009244889370165765, "epoch": 0.13150178736393944, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 8.181818181818183e-08, "loss": 0.0, "num_tokens": 7458170.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0018798112869263, "sampling/importance_sampling_ratio/mean": 1.0000689029693604, "sampling/importance_sampling_ratio/min": 0.9944168329238892, "sampling/sampling_logp_difference/max": 0.005598756484687328, "sampling/sampling_logp_difference/mean": 0.00010550412116572261, "step": 3274 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0017418517236365005, "epoch": 0.1315419528457244, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 7.87878787878788e-08, "loss": 0.0, "num_tokens": 7459834.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.002091646194458, "sampling/importance_sampling_ratio/mean": 1.0002355575561523, "sampling/importance_sampling_ratio/min": 0.9997556209564209, "sampling/sampling_logp_difference/max": 0.00208944920450449, "sampling/sampling_logp_difference/mean": 0.00023801384668331593, "step": 3275 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 69.0, "completions/max_terminated_length": 69.0, "completions/mean_length": 68.125, "completions/mean_terminated_length": 68.125, "completions/min_length": 67.0, "completions/min_terminated_length": 67.0, "entropy": 0.06052346946671605, "epoch": 0.13158211832750935, "frac_reward_zero_std": 0.0, "grad_norm": 1.703597068786621, "learning_rate": 7.575757575757576e-08, "loss": -0.0024, "num_tokens": 7461675.0, "reward": 0.999450147151947, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.999450147151947, "reward_meter_std": 0.00012958116712979972, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.000129588384879753, "reward_total_composite_mean": 0.999450147151947, "reward_total_composite_std": 0.00012958116712979972, "reward_total_mean": 0.999450147151947, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.999450147151947, "rewards/meter/std": 0.00012958116712979972, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.999450147151947, "rewards/total_composite/std": 0.00012958116712979972, "sampling/importance_sampling_ratio/max": 1.684713363647461, "sampling/importance_sampling_ratio/mean": 1.0002046823501587, "sampling/importance_sampling_ratio/min": 0.4823731780052185, "sampling/sampling_logp_difference/max": 0.7290372848510742, "sampling/sampling_logp_difference/mean": 0.011213197372853756, "step": 3276 }, { "clip_ratio/high_max": 0.018823198741301894, "clip_ratio/high_mean": 0.018823198741301894, "clip_ratio/low_mean": 0.007274207135196775, "clip_ratio/low_min": 0.007274207135196775, "clip_ratio/region_mean": 0.02609740587649867, "completions/clipped_ratio": 0.0, "completions/max_length": 308.0, "completions/max_terminated_length": 308.0, "completions/mean_length": 285.875, "completions/mean_terminated_length": 285.875, "completions/min_length": 275.0, "completions/min_terminated_length": 275.0, "entropy": 0.3647267259657383, "epoch": 0.1316222838092943, "frac_reward_zero_std": 0.0, "grad_norm": 3.1708502769470215, "learning_rate": 7.272727272727274e-08, "loss": 0.0383, "num_tokens": 7465634.0, "reward": 0.9212331175804138, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.9722222089767456, "reward_count_adherence_std": 0.05143444985151291, "reward_meter_mean": 0.9972749948501587, "reward_meter_std": 0.0005863041151314974, "reward_repeat_penalty_mean": 0.9500774145126343, "reward_repeat_penalty_std": 0.03746138885617256, "reward_std": 0.06239548325538635, "reward_total_composite_mean": 0.9212331175804138, "reward_total_composite_std": 0.062395479530096054, "reward_total_mean": 0.9212331175804138, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.9722222089767456, "rewards/count_adherence/std": 0.05143444985151291, "rewards/meter/mean": 0.9972749948501587, "rewards/meter/std": 0.0005863041151314974, "rewards/repeat_penalty/mean": 0.9500774145126343, "rewards/repeat_penalty/std": 0.03746138885617256, "rewards/total_composite/mean": 0.9212331175804138, "rewards/total_composite/std": 0.062395479530096054, "sampling/importance_sampling_ratio/max": 1.906610369682312, "sampling/importance_sampling_ratio/mean": 1.0104100704193115, "sampling/importance_sampling_ratio/min": 0.1952621191740036, "sampling/sampling_logp_difference/max": 1.6334123611450195, "sampling/sampling_logp_difference/mean": 0.039440274238586426, "step": 3277 }, { "clip_ratio/high_max": 0.03409638348966837, "clip_ratio/high_mean": 0.03409638348966837, "clip_ratio/low_mean": 0.005647590383887291, "clip_ratio/low_min": 0.005647590383887291, "clip_ratio/region_mean": 0.03974397387355566, "completions/clipped_ratio": 0.0, "completions/max_length": 337.0, "completions/max_terminated_length": 337.0, "completions/mean_length": 333.25, "completions/mean_terminated_length": 333.25, "completions/min_length": 331.0, "completions/min_terminated_length": 331.0, "entropy": 0.4740913324058056, "epoch": 0.13166244929107926, "frac_reward_zero_std": 0.0, "grad_norm": 2.639291286468506, "learning_rate": 6.96969696969697e-08, "loss": 0.0041, "num_tokens": 7469932.0, "reward": 0.9914476871490479, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9980206489562988, "reward_meter_std": 0.0009543930646032095, "reward_repeat_penalty_mean": 0.9934210777282715, "reward_repeat_penalty_std": 0.01860806532204151, "reward_std": 0.01817752793431282, "reward_total_composite_mean": 0.9914476871490479, "reward_total_composite_std": 0.01817752607166767, "reward_total_mean": 0.9914476871490479, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9980206489562988, "rewards/meter/std": 0.0009543930646032095, "rewards/repeat_penalty/mean": 0.9934210777282715, "rewards/repeat_penalty/std": 0.01860806532204151, "rewards/total_composite/mean": 0.9914476871490479, "rewards/total_composite/std": 0.01817752607166767, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.010366678237915, "sampling/importance_sampling_ratio/min": 0.14248007535934448, "sampling/sampling_logp_difference/max": 1.9485530853271484, "sampling/sampling_logp_difference/mean": 0.0520353838801384, "step": 3278 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.0007044602243695408, "epoch": 0.1317026147728642, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.666666666666668e-08, "loss": 0.0, "num_tokens": 7471564.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0012202262878418, "sampling/importance_sampling_ratio/mean": 1.0000587701797485, "sampling/importance_sampling_ratio/min": 0.9982084035873413, "sampling/sampling_logp_difference/max": 0.0017932088812813163, "sampling/sampling_logp_difference/mean": 8.241144678322598e-05, "step": 3279 }, { "clip_ratio/high_max": 0.0025510203558951616, "clip_ratio/high_mean": 0.0025510203558951616, "clip_ratio/low_mean": 0.006377550889737904, "clip_ratio/low_min": 0.006377550889737904, "clip_ratio/region_mean": 0.008928571245633066, "completions/clipped_ratio": 0.0, "completions/max_length": 98.0, "completions/max_terminated_length": 98.0, "completions/mean_length": 98.0, "completions/mean_terminated_length": 98.0, "completions/min_length": 98.0, "completions/min_terminated_length": 98.0, "entropy": 0.033778895856812596, "epoch": 0.13174278025464917, "frac_reward_zero_std": 0.0, "grad_norm": 0.038945455104112625, "learning_rate": 6.363636363636365e-08, "loss": 0.0002, "num_tokens": 7473684.0, "reward": 0.9994074702262878, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994074702262878, "reward_meter_std": 3.4306899578950834e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 3.428247964620823e-06, "reward_total_composite_mean": 0.9994074702262878, "reward_total_composite_std": 3.4306899578950834e-06, "reward_total_mean": 0.9994074702262878, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994074702262878, "rewards/meter/std": 3.4306899578950834e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994074702262878, "rewards/total_composite/std": 3.4306899578950834e-06, "sampling/importance_sampling_ratio/max": 1.345357894897461, "sampling/importance_sampling_ratio/mean": 1.0016450881958008, "sampling/importance_sampling_ratio/min": 0.7375589609146118, "sampling/sampling_logp_difference/max": 0.3044092655181885, "sampling/sampling_logp_difference/mean": 0.0032499232329428196, "step": 3280 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 34.0, "completions/max_terminated_length": 34.0, "completions/mean_length": 34.0, "completions/mean_terminated_length": 34.0, "completions/min_length": 34.0, "completions/min_terminated_length": 34.0, "entropy": 0.024989124154672027, "epoch": 0.13178294573643412, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 6.060606060606061e-08, "loss": 0.0, "num_tokens": 7475108.0, "reward": 0.9977458119392395, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9977458119392395, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9977458119392395, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9977458119392395, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9977458119392395, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9977458119392395, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0658761262893677, "sampling/importance_sampling_ratio/mean": 1.0015556812286377, "sampling/importance_sampling_ratio/min": 0.9427482485771179, "sampling/sampling_logp_difference/max": 0.06379712373018265, "sampling/sampling_logp_difference/mean": 0.002193058142438531, "step": 3281 }, { "clip_ratio/high_max": 0.0017605633474886417, "clip_ratio/high_mean": 0.0017605633474886417, "clip_ratio/low_mean": 0.0017361111240461469, "clip_ratio/low_min": 0.0017361111240461469, "clip_ratio/region_mean": 0.0034966744715347886, "completions/clipped_ratio": 0.0, "completions/max_length": 72.0, "completions/max_terminated_length": 72.0, "completions/mean_length": 71.125, "completions/mean_terminated_length": 71.125, "completions/min_length": 71.0, "completions/min_terminated_length": 71.0, "entropy": 0.0743112824857235, "epoch": 0.13182311121821907, "frac_reward_zero_std": 0.0, "grad_norm": 0.13846832513809204, "learning_rate": 5.757575757575758e-08, "loss": 0.0002, "num_tokens": 7476965.0, "reward": 0.9994271993637085, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9994271993637085, "reward_meter_std": 1.0313916391169187e-05, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.0320004548702855e-05, "reward_total_composite_mean": 0.9994271993637085, "reward_total_composite_std": 1.0313916391169187e-05, "reward_total_mean": 0.9994271993637085, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9994271993637085, "rewards/meter/std": 1.0313916391169187e-05, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9994271993637085, "rewards/total_composite/std": 1.0313916391169187e-05, "sampling/importance_sampling_ratio/max": 1.1672364473342896, "sampling/importance_sampling_ratio/mean": 1.0017709732055664, "sampling/importance_sampling_ratio/min": 0.5606830716133118, "sampling/sampling_logp_difference/max": 0.5785994529724121, "sampling/sampling_logp_difference/mean": 0.008291705511510372, "step": 3282 }, { "clip_ratio/high_max": 0.005597014795057476, "clip_ratio/high_mean": 0.005597014795057476, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.009384893695823848, "completions/clipped_ratio": 0.0, "completions/max_length": 67.0, "completions/max_terminated_length": 67.0, "completions/mean_length": 66.875, "completions/mean_terminated_length": 66.875, "completions/min_length": 66.0, "completions/min_terminated_length": 66.0, "entropy": 0.03944724937900901, "epoch": 0.13186327670000403, "frac_reward_zero_std": 0.0, "grad_norm": 0.04218384996056557, "learning_rate": 5.454545454545455e-08, "loss": -0.0006, "num_tokens": 7478764.0, "reward": 0.9981493353843689, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9981493353843689, "reward_meter_std": 8.558939043723512e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 8.561799404560588e-06, "reward_total_composite_mean": 0.9981493353843689, "reward_total_composite_std": 8.558939043723512e-06, "reward_total_mean": 0.9981493353843689, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9981493353843689, "rewards/meter/std": 8.558939043723512e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9981493353843689, "rewards/total_composite/std": 8.558939043723512e-06, "sampling/importance_sampling_ratio/max": 1.5187106132507324, "sampling/importance_sampling_ratio/mean": 1.002824306488037, "sampling/importance_sampling_ratio/min": 0.8036051392555237, "sampling/sampling_logp_difference/max": 0.4178617000579834, "sampling/sampling_logp_difference/mean": 0.005590750835835934, "step": 3283 }, { "clip_ratio/high_max": 0.0019083969527855515, "clip_ratio/high_mean": 0.0019083969527855515, "clip_ratio/low_mean": 0.0009469697251915932, "clip_ratio/low_min": 0.0009469697251915932, "clip_ratio/region_mean": 0.0028553666779771447, "completions/clipped_ratio": 0.0, "completions/max_length": 132.0, "completions/max_terminated_length": 132.0, "completions/mean_length": 131.75, "completions/mean_terminated_length": 131.75, "completions/min_length": 131.0, "completions/min_terminated_length": 131.0, "entropy": 0.08821968082338572, "epoch": 0.13190344218178898, "frac_reward_zero_std": 0.0, "grad_norm": 1.1970688104629517, "learning_rate": 5.151515151515152e-08, "loss": -0.0012, "num_tokens": 7481306.0, "reward": 0.9992576837539673, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992576837539673, "reward_meter_std": 0.00048146978951990604, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0004814636486116797, "reward_total_composite_mean": 0.9992576837539673, "reward_total_composite_std": 0.00048146978951990604, "reward_total_mean": 0.9992576837539673, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992576837539673, "rewards/meter/std": 0.00048146978951990604, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9992576837539673, "rewards/total_composite/std": 0.00048146978951990604, "sampling/importance_sampling_ratio/max": 1.4763288497924805, "sampling/importance_sampling_ratio/mean": 1.0037535429000854, "sampling/importance_sampling_ratio/min": 0.4606848359107971, "sampling/sampling_logp_difference/max": 0.7750411033630371, "sampling/sampling_logp_difference/mean": 0.009545176289975643, "step": 3284 }, { "clip_ratio/high_max": 0.005871425149962306, "clip_ratio/high_mean": 0.005871425149962306, "clip_ratio/low_mean": 0.01729804463684559, "clip_ratio/low_min": 0.01729804463684559, "clip_ratio/region_mean": 0.023169469786807895, "completions/clipped_ratio": 0.0, "completions/max_length": 385.0, "completions/max_terminated_length": 385.0, "completions/mean_length": 361.125, "completions/mean_terminated_length": 361.125, "completions/min_length": 347.0, "completions/min_terminated_length": 347.0, "entropy": 0.4625861681997776, "epoch": 0.13194360766357394, "frac_reward_zero_std": 0.0, "grad_norm": 1.6411669254302979, "learning_rate": 4.8484848484848486e-08, "loss": -0.0227, "num_tokens": 7485883.0, "reward": 0.827822208404541, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8409091234207153, "reward_count_adherence_std": 0.04208274558186531, "reward_meter_mean": 0.9987712502479553, "reward_meter_std": 0.0013375915586948395, "reward_repeat_penalty_mean": 0.9852941036224365, "reward_repeat_penalty_std": 0.027230001986026764, "reward_std": 0.053048938512802124, "reward_total_composite_mean": 0.827822208404541, "reward_total_composite_std": 0.05304892361164093, "reward_total_mean": 0.827822208404541, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8409091234207153, "rewards/count_adherence/std": 0.04208274558186531, "rewards/meter/mean": 0.9987712502479553, "rewards/meter/std": 0.0013375915586948395, "rewards/repeat_penalty/mean": 0.9852941036224365, "rewards/repeat_penalty/std": 0.027230001986026764, "rewards/total_composite/mean": 0.827822208404541, "rewards/total_composite/std": 0.05304892361164093, "sampling/importance_sampling_ratio/max": 1.6690725088119507, "sampling/importance_sampling_ratio/mean": 1.01465904712677, "sampling/importance_sampling_ratio/min": 0.18927277624607086, "sampling/sampling_logp_difference/max": 1.6645660400390625, "sampling/sampling_logp_difference/mean": 0.04496872425079346, "step": 3285 }, { "clip_ratio/high_max": 0.008415504998993129, "clip_ratio/high_mean": 0.008415504998993129, "clip_ratio/low_mean": 0.0021067415946163237, "clip_ratio/low_min": 0.0021067415946163237, "clip_ratio/region_mean": 0.010522246593609452, "completions/clipped_ratio": 0.0, "completions/max_length": 180.0, "completions/max_terminated_length": 180.0, "completions/mean_length": 178.125, "completions/mean_terminated_length": 178.125, "completions/min_length": 177.0, "completions/min_terminated_length": 177.0, "entropy": 0.24275393225252628, "epoch": 0.1319837731453589, "frac_reward_zero_std": 0.0, "grad_norm": 0.6697944402694702, "learning_rate": 4.545454545454546e-08, "loss": 0.0016, "num_tokens": 7488852.0, "reward": 0.9990102052688599, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9990102052688599, "reward_meter_std": 0.00010323245805921033, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0001032423970173113, "reward_total_composite_mean": 0.9990102052688599, "reward_total_composite_std": 0.00010323245805921033, "reward_total_mean": 0.9990102052688599, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9990102052688599, "rewards/meter/std": 0.00010323245805921033, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9990102052688599, "rewards/total_composite/std": 0.00010323245805921033, "sampling/importance_sampling_ratio/max": 1.7193830013275146, "sampling/importance_sampling_ratio/mean": 1.006458044052124, "sampling/importance_sampling_ratio/min": 0.44476625323295593, "sampling/sampling_logp_difference/max": 0.810206413269043, "sampling/sampling_logp_difference/mean": 0.02063995786011219, "step": 3286 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0020569458283716813, "epoch": 0.13202393862714384, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 4.2424242424242435e-08, "loss": 0.0, "num_tokens": 7490484.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.0030535459518433, "sampling/importance_sampling_ratio/mean": 1.0002548694610596, "sampling/importance_sampling_ratio/min": 0.9995039105415344, "sampling/sampling_logp_difference/max": 0.0030488455668091774, "sampling/sampling_logp_difference/mean": 0.0002587019116617739, "step": 3287 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 64.0, "completions/max_terminated_length": 64.0, "completions/mean_length": 64.0, "completions/mean_terminated_length": 64.0, "completions/min_length": 64.0, "completions/min_terminated_length": 64.0, "entropy": 0.001106615709431935, "epoch": 0.1320641041089288, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 3.93939393939394e-08, "loss": 0.0, "num_tokens": 7492204.0, "reward": 0.9993994235992432, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9993994235992432, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.9993994235992432, "reward_total_composite_std": 0.0, "reward_total_mean": 0.9993994235992432, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9993994235992432, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9993994235992432, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.002815842628479, "sampling/importance_sampling_ratio/mean": 1.0000782012939453, "sampling/importance_sampling_ratio/min": 0.9984528422355652, "sampling/sampling_logp_difference/max": 0.0028118479531258345, "sampling/sampling_logp_difference/mean": 9.810684423428029e-05, "step": 3288 }, { "clip_ratio/high_max": 0.025478466413915157, "clip_ratio/high_mean": 0.025478466413915157, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.025478466413915157, "completions/clipped_ratio": 0.0, "completions/max_length": 108.0, "completions/max_terminated_length": 108.0, "completions/mean_length": 103.625, "completions/mean_terminated_length": 103.625, "completions/min_length": 101.0, "completions/min_terminated_length": 101.0, "entropy": 0.2910528890788555, "epoch": 0.13210426959071375, "frac_reward_zero_std": 0.0, "grad_norm": 3.059532642364502, "learning_rate": 3.636363636363637e-08, "loss": 0.0109, "num_tokens": 7494281.0, "reward": 0.9677305221557617, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9926178455352783, "reward_meter_std": 0.0021126074716448784, "reward_repeat_penalty_mean": 0.9750000238418579, "reward_repeat_penalty_std": 0.0707106739282608, "reward_std": 0.06925328075885773, "reward_total_composite_mean": 0.9677305221557617, "reward_total_composite_std": 0.06925326585769653, "reward_total_mean": 0.9677305221557617, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9926178455352783, "rewards/meter/std": 0.0021126074716448784, "rewards/repeat_penalty/mean": 0.9750000238418579, "rewards/repeat_penalty/std": 0.0707106739282608, "rewards/total_composite/mean": 0.9677305221557617, "rewards/total_composite/std": 0.06925326585769653, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.0063260793685913, "sampling/importance_sampling_ratio/min": 0.2426658272743225, "sampling/sampling_logp_difference/max": 1.4160699844360352, "sampling/sampling_logp_difference/mean": 0.03573079779744148, "step": 3289 }, { "clip_ratio/high_max": 0.025775008834898472, "clip_ratio/high_mean": 0.025775008834898472, "clip_ratio/low_mean": 0.009898273507133126, "clip_ratio/low_min": 0.009898273507133126, "clip_ratio/region_mean": 0.0356732823420316, "completions/clipped_ratio": 0.0, "completions/max_length": 209.0, "completions/max_terminated_length": 209.0, "completions/mean_length": 203.5, "completions/mean_terminated_length": 203.5, "completions/min_length": 201.0, "completions/min_terminated_length": 201.0, "entropy": 0.38689322397112846, "epoch": 0.1321444350724987, "frac_reward_zero_std": 0.0, "grad_norm": 2.9757983684539795, "learning_rate": 3.333333333333334e-08, "loss": 0.0003, "num_tokens": 7497469.0, "reward": 0.9552866816520691, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9778822660446167, "reward_meter_std": 0.03634575754404068, "reward_repeat_penalty_mean": 0.9772727489471436, "reward_repeat_penalty_std": 0.04208271950483322, "reward_std": 0.047207050025463104, "reward_total_composite_mean": 0.9552866816520691, "reward_total_composite_std": 0.0472070537507534, "reward_total_mean": 0.9552866816520691, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9778822660446167, "rewards/meter/std": 0.03634575754404068, "rewards/repeat_penalty/mean": 0.9772727489471436, "rewards/repeat_penalty/std": 0.04208271950483322, "rewards/total_composite/mean": 0.9552866816520691, "rewards/total_composite/std": 0.0472070537507534, "sampling/importance_sampling_ratio/max": 1.8156616687774658, "sampling/importance_sampling_ratio/mean": 1.0083707571029663, "sampling/importance_sampling_ratio/min": 0.29560860991477966, "sampling/sampling_logp_difference/max": 1.2187190055847168, "sampling/sampling_logp_difference/mean": 0.038450971245765686, "step": 3290 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.005710955825634301, "clip_ratio/low_min": 0.005710955825634301, "clip_ratio/region_mean": 0.005710955825634301, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.625, "completions/mean_terminated_length": 65.625, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.06898282328620553, "epoch": 0.13218460055428366, "frac_reward_zero_std": 0.0, "grad_norm": 2.9630963802337646, "learning_rate": 3.0303030303030305e-08, "loss": 0.0059, "num_tokens": 7499282.0, "reward": 0.9937740564346313, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9937740564346313, "reward_meter_std": 0.0003091543912887573, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0003091563412453979, "reward_total_composite_mean": 0.9937740564346313, "reward_total_composite_std": 0.0003091543912887573, "reward_total_mean": 0.9937740564346313, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9937740564346313, "rewards/meter/std": 0.0003091543912887573, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9937740564346313, "rewards/total_composite/std": 0.0003091543912887573, "sampling/importance_sampling_ratio/max": 1.6661887168884277, "sampling/importance_sampling_ratio/mean": 1.0025181770324707, "sampling/importance_sampling_ratio/min": 0.37397557497024536, "sampling/sampling_logp_difference/max": 0.9835647940635681, "sampling/sampling_logp_difference/mean": 0.01210538949817419, "step": 3291 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0016719198902137578, "epoch": 0.13222476603606861, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 2.7272727272727276e-08, "loss": 0.0, "num_tokens": 7501162.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.001773476600647, "sampling/importance_sampling_ratio/mean": 1.000186800956726, "sampling/importance_sampling_ratio/min": 0.9987837076187134, "sampling/sampling_logp_difference/max": 0.0017719701863825321, "sampling/sampling_logp_difference/mean": 0.000192559469724074, "step": 3292 }, { "clip_ratio/high_max": 0.014237138908356428, "clip_ratio/high_mean": 0.014237138908356428, "clip_ratio/low_mean": 0.008654525852762163, "clip_ratio/low_min": 0.008654525852762163, "clip_ratio/region_mean": 0.02289166476111859, "completions/clipped_ratio": 0.0, "completions/max_length": 448.0, "completions/max_terminated_length": 448.0, "completions/mean_length": 428.0, "completions/mean_terminated_length": 428.0, "completions/min_length": 408.0, "completions/min_terminated_length": 408.0, "entropy": 0.39450087025761604, "epoch": 0.13226493151785357, "frac_reward_zero_std": 0.0, "grad_norm": 1.2802501916885376, "learning_rate": 2.4242424242424243e-08, "loss": -0.0169, "num_tokens": 7506578.0, "reward": 0.8016811609268188, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 0.8392857313156128, "reward_count_adherence_std": 0.03306501731276512, "reward_meter_mean": 0.9988534450531006, "reward_meter_std": 0.00014253715926315635, "reward_repeat_penalty_mean": 0.9562541246414185, "reward_repeat_penalty_std": 0.03288932517170906, "reward_std": 0.042949263006448746, "reward_total_composite_mean": 0.8016811609268188, "reward_total_composite_std": 0.04294924437999725, "reward_total_mean": 0.8016811609268188, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 0.8392857313156128, "rewards/count_adherence/std": 0.03306501731276512, "rewards/meter/mean": 0.9988534450531006, "rewards/meter/std": 0.00014253715926315635, "rewards/repeat_penalty/mean": 0.9562541246414185, "rewards/repeat_penalty/std": 0.03288932517170906, "rewards/total_composite/mean": 0.8016811609268188, "rewards/total_composite/std": 0.04294924437999725, "sampling/importance_sampling_ratio/max": 1.804500699043274, "sampling/importance_sampling_ratio/mean": 1.0091326236724854, "sampling/importance_sampling_ratio/min": 0.19196099042892456, "sampling/sampling_logp_difference/max": 1.6504631042480469, "sampling/sampling_logp_difference/mean": 0.03670765459537506, "step": 3293 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 61.0, "completions/max_terminated_length": 61.0, "completions/mean_length": 61.0, "completions/mean_terminated_length": 61.0, "completions/min_length": 61.0, "completions/min_terminated_length": 61.0, "entropy": 0.012461528764106333, "epoch": 0.13230509699963852, "frac_reward_zero_std": 0.0, "grad_norm": 0.030057502910494804, "learning_rate": 2.1212121212121217e-08, "loss": -0.0, "num_tokens": 7508354.0, "reward": 0.9973385334014893, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9973385334014893, "reward_meter_std": 1.0115243185282452e-06, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 1.0115243185282452e-06, "reward_total_composite_mean": 0.9973385334014893, "reward_total_composite_std": 1.0115243185282452e-06, "reward_total_mean": 0.9973385334014893, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9973385334014893, "rewards/meter/std": 1.0115243185282452e-06, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9973385334014893, "rewards/total_composite/std": 1.0115243185282452e-06, "sampling/importance_sampling_ratio/max": 1.0437626838684082, "sampling/importance_sampling_ratio/mean": 0.9997027516365051, "sampling/importance_sampling_ratio/min": 0.5063959360122681, "sampling/sampling_logp_difference/max": 0.680436372756958, "sampling/sampling_logp_difference/mean": 0.002317759208381176, "step": 3294 }, { "clip_ratio/high_max": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/region_mean": 0.0, "completions/clipped_ratio": 0.0, "completions/max_length": 54.0, "completions/max_terminated_length": 54.0, "completions/mean_length": 54.0, "completions/mean_terminated_length": 54.0, "completions/min_length": 54.0, "completions/min_terminated_length": 54.0, "entropy": 0.0021129547676537186, "epoch": 0.13234526248142348, "frac_reward_zero_std": 1.0, "grad_norm": 0.0, "learning_rate": 1.8181818181818185e-08, "loss": 0.0, "num_tokens": 7510050.0, "reward": 0.787638783454895, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.787638783454895, "reward_meter_std": 0.0, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0, "reward_total_composite_mean": 0.787638783454895, "reward_total_composite_std": 0.0, "reward_total_mean": 0.787638783454895, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.787638783454895, "rewards/meter/std": 0.0, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.787638783454895, "rewards/total_composite/std": 0.0, "sampling/importance_sampling_ratio/max": 1.002079963684082, "sampling/importance_sampling_ratio/mean": 1.000232458114624, "sampling/importance_sampling_ratio/min": 0.9994556307792664, "sampling/sampling_logp_difference/max": 0.002077840268611908, "sampling/sampling_logp_difference/mean": 0.00023673544637858868, "step": 3295 }, { "clip_ratio/high_max": 0.012068168085534126, "clip_ratio/high_mean": 0.012068168085534126, "clip_ratio/low_mean": 0.0025125627871602774, "clip_ratio/low_min": 0.0025125627871602774, "clip_ratio/region_mean": 0.014580730872694403, "completions/clipped_ratio": 0.0, "completions/max_length": 200.0, "completions/max_terminated_length": 200.0, "completions/mean_length": 197.125, "completions/mean_terminated_length": 197.125, "completions/min_length": 192.0, "completions/min_terminated_length": 192.0, "entropy": 0.2336829099804163, "epoch": 0.13238542796320843, "frac_reward_zero_std": 0.0, "grad_norm": 1.9730912446975708, "learning_rate": 1.5151515151515152e-08, "loss": 0.0072, "num_tokens": 7513219.0, "reward": 0.9879227876663208, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9992784261703491, "reward_meter_std": 0.00010585635027382523, "reward_repeat_penalty_mean": 0.9886363744735718, "reward_repeat_penalty_std": 0.03214120864868164, "reward_std": 0.032112736254930496, "reward_total_composite_mean": 0.9879227876663208, "reward_total_composite_std": 0.0321127250790596, "reward_total_mean": 0.9879227876663208, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9992784261703491, "rewards/meter/std": 0.00010585635027382523, "rewards/repeat_penalty/mean": 0.9886363744735718, "rewards/repeat_penalty/std": 0.03214120864868164, "rewards/total_composite/mean": 0.9879227876663208, "rewards/total_composite/std": 0.0321127250790596, "sampling/importance_sampling_ratio/max": 1.8454686403274536, "sampling/importance_sampling_ratio/mean": 1.009697675704956, "sampling/importance_sampling_ratio/min": 0.40942853689193726, "sampling/sampling_logp_difference/max": 0.8929929733276367, "sampling/sampling_logp_difference/mean": 0.021909546107053757, "step": 3296 }, { "clip_ratio/high_max": 0.03054804727435112, "clip_ratio/high_mean": 0.03054804727435112, "clip_ratio/low_mean": 0.010391832329332829, "clip_ratio/low_min": 0.010391832329332829, "clip_ratio/region_mean": 0.04093987960368395, "completions/clipped_ratio": 0.0, "completions/max_length": 316.0, "completions/max_terminated_length": 316.0, "completions/mean_length": 302.25, "completions/mean_terminated_length": 302.25, "completions/min_length": 291.0, "completions/min_terminated_length": 291.0, "entropy": 0.5184807442128658, "epoch": 0.13242559344499338, "frac_reward_zero_std": 0.0, "grad_norm": 2.3195743560791016, "learning_rate": 1.2121212121212122e-08, "loss": -0.0017, "num_tokens": 7517109.0, "reward": 0.9883631467819214, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9957056641578674, "reward_meter_std": 0.006368847563862801, "reward_repeat_penalty_mean": 0.9926470518112183, "reward_repeat_penalty_std": 0.020797256380319595, "reward_std": 0.0205837395042181, "reward_total_composite_mean": 0.9883631467819214, "reward_total_composite_std": 0.020583728328347206, "reward_total_mean": 0.9883631467819214, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9957056641578674, "rewards/meter/std": 0.006368847563862801, "rewards/repeat_penalty/mean": 0.9926470518112183, "rewards/repeat_penalty/std": 0.020797256380319595, "rewards/total_composite/mean": 0.9883631467819214, "rewards/total_composite/std": 0.020583728328347206, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.014825701713562, "sampling/importance_sampling_ratio/min": 0.3243544399738312, "sampling/sampling_logp_difference/max": 1.1259183883666992, "sampling/sampling_logp_difference/mean": 0.05743308737874031, "step": 3297 }, { "clip_ratio/high_max": 0.005710955825634301, "clip_ratio/high_mean": 0.005710955825634301, "clip_ratio/low_mean": 0.0037878789007663727, "clip_ratio/low_min": 0.0037878789007663727, "clip_ratio/region_mean": 0.009498834726400673, "completions/clipped_ratio": 0.0, "completions/max_length": 66.0, "completions/max_terminated_length": 66.0, "completions/mean_length": 65.375, "completions/mean_terminated_length": 65.375, "completions/min_length": 65.0, "completions/min_terminated_length": 65.0, "entropy": 0.11512181628495455, "epoch": 0.13246575892677834, "frac_reward_zero_std": 0.0, "grad_norm": 2.183478593826294, "learning_rate": 9.090909090909092e-09, "loss": 0.0017, "num_tokens": 7518824.0, "reward": 0.9922501444816589, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9922501444816589, "reward_meter_std": 0.004244488663971424, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0042444937862455845, "reward_total_composite_mean": 0.9922501444816589, "reward_total_composite_std": 0.004244488663971424, "reward_total_mean": 0.9922501444816589, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9922501444816589, "rewards/meter/std": 0.004244488663971424, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9922501444816589, "rewards/total_composite/std": 0.004244488663971424, "sampling/importance_sampling_ratio/max": 1.4506497383117676, "sampling/importance_sampling_ratio/mean": 1.0067819356918335, "sampling/importance_sampling_ratio/min": 0.49525120854377747, "sampling/sampling_logp_difference/max": 0.7026901245117188, "sampling/sampling_logp_difference/mean": 0.013568882830440998, "step": 3298 }, { "clip_ratio/high_max": 0.015017259865999222, "clip_ratio/high_mean": 0.015017259865999222, "clip_ratio/low_mean": 0.01500677247531712, "clip_ratio/low_min": 0.01500677247531712, "clip_ratio/region_mean": 0.030024032341316342, "completions/clipped_ratio": 0.0, "completions/max_length": 170.0, "completions/max_terminated_length": 170.0, "completions/mean_length": 166.5, "completions/mean_terminated_length": 166.5, "completions/min_length": 159.0, "completions/min_terminated_length": 159.0, "entropy": 0.3029062431305647, "epoch": 0.1325059244085633, "frac_reward_zero_std": 0.0, "grad_norm": 3.7170727252960205, "learning_rate": 6.060606060606061e-09, "loss": 0.0057, "num_tokens": 7521612.0, "reward": 0.9987373352050781, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9987373352050781, "reward_meter_std": 0.000354176911059767, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.00035417175968177617, "reward_total_composite_mean": 0.9987373352050781, "reward_total_composite_std": 0.000354176911059767, "reward_total_mean": 0.9987373352050781, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9987373352050781, "rewards/meter/std": 0.000354176911059767, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9987373352050781, "rewards/total_composite/std": 0.000354176911059767, "sampling/importance_sampling_ratio/max": 2.0, "sampling/importance_sampling_ratio/mean": 1.001840591430664, "sampling/importance_sampling_ratio/min": 0.13569968938827515, "sampling/sampling_logp_difference/max": 1.9973111152648926, "sampling/sampling_logp_difference/mean": 0.03916100785136223, "step": 3299 }, { "clip_ratio/high_max": 0.0027173913549631834, "clip_ratio/high_mean": 0.0027173913549631834, "clip_ratio/low_mean": 0.00815217406488955, "clip_ratio/low_min": 0.00815217406488955, "clip_ratio/region_mean": 0.010869565419852734, "completions/clipped_ratio": 0.0, "completions/max_length": 46.0, "completions/max_terminated_length": 46.0, "completions/mean_length": 46.0, "completions/mean_terminated_length": 46.0, "completions/min_length": 46.0, "completions/min_terminated_length": 46.0, "entropy": 0.09115941543132067, "epoch": 0.13254608989034825, "frac_reward_zero_std": 0.0, "grad_norm": 6.677674293518066, "learning_rate": 3.0303030303030304e-09, "loss": 0.0153, "num_tokens": 7523188.0, "reward": 0.9442145824432373, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_meter_mean": 0.9442145824432373, "reward_meter_std": 0.003297017654404044, "reward_repeat_penalty_mean": 1.0, "reward_repeat_penalty_std": 0.0, "reward_std": 0.0032970213796943426, "reward_total_composite_mean": 0.9442145824432373, "reward_total_composite_std": 0.003297017654404044, "reward_total_mean": 0.9442145824432373, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/meter/mean": 0.9442145824432373, "rewards/meter/std": 0.003297017654404044, "rewards/repeat_penalty/mean": 1.0, "rewards/repeat_penalty/std": 0.0, "rewards/total_composite/mean": 0.9442145824432373, "rewards/total_composite/std": 0.003297017654404044, "sampling/importance_sampling_ratio/max": 1.663463830947876, "sampling/importance_sampling_ratio/mean": 1.0027215480804443, "sampling/importance_sampling_ratio/min": 0.11092793941497803, "sampling/sampling_logp_difference/max": 2.1988744735717773, "sampling/sampling_logp_difference/mean": 0.021698685362935066, "step": 3300 }, { "epoch": 0.13254608989034825, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_completions/clipped_ratio": 0.009615384615384616, "eval_completions/max_length": 421.38461538461536, "eval_completions/max_terminated_length": 412.2307692307692, "eval_completions/mean_length": 213.14423076923077, "eval_completions/mean_terminated_length": 209.98489027756912, "eval_completions/min_length": 61.23076923076923, "eval_completions/min_terminated_length": 61.23076923076923, "eval_entropy": 0.38235602699793303, "eval_frac_reward_zero_std": 0.0, "eval_loss": NaN, "eval_num_tokens": 7523188.0, "eval_reward": 0.7239617155148432, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_count_adherence_mean": 0.9621203220807589, "eval_reward_count_adherence_std": 0.06154714152216911, "eval_reward_meter_mean": 0.7804106657321637, "eval_reward_meter_std": 0.3429693900621854, "eval_reward_repeat_penalty_mean": 0.9574309633328364, "eval_reward_repeat_penalty_std": 0.07397974411455485, "eval_reward_std": NaN, "eval_reward_total_composite_mean": 0.7239617155148432, "eval_reward_total_composite_std": 0.3431667788670613, "eval_reward_total_mean": 0.7239617155148432, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/count_adherence/mean": 0.9621203220807589, "eval_rewards/count_adherence/std": 0.06154714152216911, "eval_rewards/meter/mean": 0.7804106657321637, "eval_rewards/meter/std": 0.3429693900621854, "eval_rewards/repeat_penalty/mean": 0.9574309633328364, "eval_rewards/repeat_penalty/std": 0.07397974411455485, "eval_rewards/total_composite/mean": 0.7239617155148432, "eval_rewards/total_composite/std": 0.3431667788670613, "eval_runtime": 80.3156, "eval_samples_per_second": 1.295, "eval_sampling/importance_sampling_ratio/max": 1.6007587084403405, "eval_sampling/importance_sampling_ratio/mean": 1.0095902956449068, "eval_sampling/importance_sampling_ratio/min": 0.2969068231490942, "eval_sampling/sampling_logp_difference/max": 1.2331663278432994, "eval_sampling/sampling_logp_difference/mean": 0.03249364231641476, "eval_steps_per_second": 0.162, "step": 3300 } ], "logging_steps": 1, "max_steps": 3300, "num_input_tokens_seen": 7523188, "num_train_epochs": 1, "save_steps": 50, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 0.0, "train_batch_size": 1, "trial_name": null, "trial_params": null }